From f18fe20d9e8b10f4793ba5a854cf09dff8ddce62 Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Sat, 26 Sep 2026 10:31:39 +0200 Subject: [PATCH 01/18] Move the child plumbing out of py_isolated The socket listen/accept sequence, frames, executable lookup and port options move to py_child, so another process can drive a Python child the same way. No change in behaviour. --- docs/code-map.md | 1 + src/py_child.erl | 260 ++++++++++++++++++++++++++++++++++++++++++++ src/py_isolated.erl | 203 ++++------------------------------ 3 files changed, 285 insertions(+), 179 deletions(-) create mode 100644 src/py_child.erl diff --git a/docs/code-map.md b/docs/code-map.md index 9da9c9b..905d055 100644 --- a/docs/code-map.md +++ b/docs/code-map.md @@ -16,6 +16,7 @@ exercised by suites). Guides are in `docs/`, suites in `test/`. Start with | `py_context` | The API every mode answers (`call/eval/exec`, `interrupt`, `kill`, loops, `pass_fd`), the reply protocol and the pid to NIF reference table; `init/4` hands the process to `py_context_embedded` or `py_isolated` | live | context-affinity, workers, interrupts | `py_context_SUITE`, `py_context_process_SUITE`, `py_interrupt_SUITE`, `py_worker_loop_SUITE` | | `py_context_embedded` | Process body for `worker` and `owngil` mode: the receive loop, callbacks (suspension and pipe), worker loops | live | architecture, state-machines | same | | `py_isolated` | `gen_statem` driving a child process over the socket; restart policy | live | isolated | `py_isolated_*_SUITE` | +| `py_child` | Helpers shared by the processes that drive a Python child: executable lookup, socket listen/accept, frames, port env, rlimit flags | live | isolated | `py_isolated_*_SUITE` | | `py_context_router` | Pools and scheduler-affinity routing | live | pools, context-affinity | `py_context_router_SUITE`, `py_pool_SUITE` | | `py_context_sup`, `py_context_init` | Supervisor of contexts; starts the default pool at boot | live | pools | (through the above) | | `py_nif` | Erlang stubs and docs for every NIF | live | api-reference | all | diff --git a/src/py_child.erl b/src/py_child.erl new file mode 100644 index 0000000..47add4b --- /dev/null +++ b/src/py_child.erl @@ -0,0 +1,260 @@ +%% Copyright 2026 Benoit Chesneau +%% +%% Licensed under the Apache License, Version 2.0 (the "License"); +%% you may not use this file except in compliance with the License. +%% You may obtain a copy of the License at +%% +%% http://www.apache.org/licenses/LICENSE-2.0 +%% +%% Unless required by applicable law or agreed to in writing, software +%% distributed under the License is distributed on an "AS IS" BASIS, +%% WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +%% See the License for the specific language governing permissions and +%% limitations under the License. + +%%% @doc Plumbing shared by the processes that drive a Python child over a +%%% Unix socket: `py_isolated' (one context, one child) and +%%% `py_session_template' (the zygote sessions are forked from). +%%% +%%% Frames are `<>', +%%% the format of the blocking callback pipe. +%%% +%%% @private +%%% +%%% Owns: nothing; pure helpers and the listen/accept sequence. +%%% Never: keeps state between calls. +-module(py_child). + +-export([python_executable/1, + priv_dir/0, + sock_dir/0, + new_sock_path/1, + listen/1, + accept/3, + tune_socket/1, + port_env/1, + rlimit_args/1, + cgroup_args/1, + frame/3, + send_frame/4, + parse_frame/1, + exit_reason/1, + kill_os_pid/1, + to_bin/1, + to_list/1]). + +-define(SOCKET_BUF, 1024 * 1024). + +%% @doc Python executable used for children: the `python' option, then the +%% `isolated_python' application env, then the interpreter matching the +%% embedded runtime, then `python3' from PATH. +-spec python_executable(map()) -> string() | {error, term()}. +python_executable(Opts) -> + Candidate = case maps:get(python, Opts, undefined) of + undefined -> + case application:get_env(erlang_python, isolated_python) of + {ok, P} -> P; + undefined -> default_python() + end; + P -> P + end, + resolve_exe(to_list(Candidate)). + +default_python() -> + case persistent_term:get({?MODULE, python}, undefined) of + undefined -> + Exe = try py:python_executable() catch _:_ -> "python3" end, + persistent_term:put({?MODULE, python}, Exe), + Exe; + Exe -> + Exe + end. + +resolve_exe(Exe) -> + case filename:pathtype(Exe) of + absolute -> + case filelib:is_file(Exe) of + true -> Exe; + false -> {error, {python_not_found, Exe}} + end; + _ -> + case os:find_executable(Exe) of + false -> {error, {python_not_found, Exe}}; + Found -> Found + end + end. + +-spec priv_dir() -> file:filename(). +priv_dir() -> + case code:priv_dir(erlang_python) of + {error, bad_name} -> + filename:join(filename:dirname(filename:dirname(code:which(?MODULE))), "priv"); + Dir -> + Dir + end. + +%% @doc Private directory for the sockets of this node (mode 0700). Kept +%% under `$TMPDIR': a Unix socket path is limited to 104 bytes. +-spec sock_dir() -> file:filename(). +sock_dir() -> + Base = case os:getenv("TMPDIR") of + false -> "/tmp"; + T -> T + end, + Dir = filename:join(Base, "erlang_python_" ++ os:getpid()), + ok = filelib:ensure_dir(filename:join(Dir, "x")), + _ = file:change_mode(Dir, 8#700), + Dir. + +-spec new_sock_path(string()) -> file:filename(). +new_sock_path(Prefix) -> + filename:join(sock_dir(), + Prefix ++ integer_to_list(erlang:unique_integer([positive])) ++ ".sock"). + +%% @doc Listening socket at Path, for one child to connect to. +-spec listen(file:filename()) -> {ok, socket:socket()} | {error, term()}. +listen(Path) -> + _ = file:delete(Path), + case socket:open(local, stream, default) of + {ok, L} -> + case socket:bind(L, #{family => local, path => Path}) of + ok -> + case socket:listen(L) of + ok -> {ok, L}; + {error, _} = Err -> socket:close(L), Err + end; + {error, _} = Err -> + socket:close(L), Err + end; + {error, Reason} -> + {error, {socket_open_failed, Reason}} + end. + +%% @doc Accept the child's connection while watching its port, so a child +%% that dies before connecting (bad interpreter, missing script) is reported +%% with its output instead of timing out. +-spec accept(socket:socket(), {port, port()}, timeout()) -> + {ok, socket:socket()} | {error, term()}. +accept(L, Watch, Timeout) -> + Deadline = erlang:monotonic_time(millisecond) + Timeout, + accept(L, Watch, Deadline, []). + +accept(L, Watch, Deadline, Out) -> + case socket:accept(L, nowait) of + {ok, S} -> + %% Output printed before connecting is still worth logging + {port, Port} = Watch, + [self() ! {Port, {data, D}} || D <- lists:reverse(Out)], + {ok, S}; + {select, {select_info, _, Handle}} -> + Left = max(0, Deadline - erlang:monotonic_time(millisecond)), + receive + {'$socket', L, select, Handle} -> + accept(L, Watch, Deadline, Out); + {Port, {exit_status, Status}} when Watch =:= {port, Port} -> + _ = socket:cancel(L, {select_info, accept, Handle}), + {error, {child_exited_at_start, exit_reason(Status), + drain_port_output(Port, Out)}}; + {Port, {data, D}} when Watch =:= {port, Port} -> + %% Keep it here, not in the mailbox: re-sending it would + %% make this receive return at once and never time out + accept(L, Watch, Deadline, [D | Out]) + after Left -> + {port, Port} = Watch, + _ = socket:cancel(L, {select_info, accept, Handle}), + {error, {start_timeout, drain_port_output(Port, Out)}} + end; + {error, Reason} -> + {error, {accept_failed, Reason}} + end. + +drain_port_output(Port, Acc) -> + receive + {Port, {data, D}} -> drain_port_output(Port, [D | Acc]) + after 50 -> + iolist_to_binary(lists:reverse(Acc)) + end. + +%% @doc Default Unix socket buffers are small (8 KB on macOS); large payloads +%% would cross in hundreds of wakeups. Best effort: the kernel clamps. +-spec tune_socket(socket:socket()) -> ok. +tune_socket(S) -> + _ = socket:setopt(S, {otp, rcvbuf}, ?SOCKET_BUF), + _ = socket:setopt(S, {socket, rcvbuf}, ?SOCKET_BUF), + _ = socket:setopt(S, {socket, sndbuf}, ?SOCKET_BUF), + ok. + +%% @doc The `{env, ...}' port option for a child: `env' adds variables to +%% what the child inherits from the VM. +-spec port_env(map()) -> [{string(), string() | false}]. +port_env(Opts) -> + [{to_list(K), to_list(V)} || {K, V} <- maps:to_list(maps:get(env, Opts, #{}))]. + +-spec rlimit_args(map()) -> [string()]. +rlimit_args(Opts) -> + Limits = maps:get(rlimits, Opts, #{}), + lists:append([case maps:get(K, Limits, undefined) of + undefined -> []; + V when is_integer(V), V >= 0 -> ["--rlimit-" ++ atom_to_list(K), integer_to_list(V)] + end || K <- [as, cpu, nofile]]). + +-spec cgroup_args(map()) -> [string()]. +cgroup_args(Opts) -> + case maps:get(cgroup, Opts, undefined) of + undefined -> []; + Dir -> ["--cgroup", to_list(Dir)] + end. + +-spec frame(non_neg_integer(), byte(), binary()) -> binary(). +frame(Id, Status, Payload) -> + Body = <>, + <>. + +-spec send_frame(socket:socket(), non_neg_integer(), byte(), term()) -> ok | {error, term()}. +send_frame(S, Id, Status, Term) -> + case socket:send(S, frame(Id, Status, term_to_binary(Term))) of + ok -> ok; + {error, {Reason, _Rest}} -> {error, Reason}; + {error, Reason} -> {error, Reason} + end. + +-spec parse_frame(binary()) -> + {ok, {non_neg_integer(), byte(), term()}, binary()} | more | {error, term()}. +parse_frame(<>) -> + case Body of + <> -> + try + Term = case Payload of + <<>> -> undefined; + _ -> binary_to_term(Payload) + end, + {ok, {Id, Status, Term}, Rest} + catch + error:badarg -> {error, bad_etf} + end; + <<>> -> + {error, empty_body} + end; +parse_frame(_) -> + more. + +%% @doc Exit reason from a port exit status (128 + N is signal N). +-spec exit_reason(integer()) -> {signal, pos_integer()} | {exit_status, integer()}. +exit_reason(Status) when Status > 128 -> {signal, Status - 128}; +exit_reason(Status) -> {exit_status, Status}. + +-spec kill_os_pid(integer()) -> ok. +kill_os_pid(OsPid) when is_integer(OsPid), OsPid > 0 -> + _ = py_nif:os_kill(OsPid, 9), + ok; +kill_os_pid(_) -> + ok. + +to_bin(A) when is_atom(A) -> atom_to_binary(A, utf8); +to_bin(L) when is_list(L) -> unicode:characters_to_binary(L); +to_bin(B) when is_binary(B) -> B. + +to_list(A) when is_atom(A) -> atom_to_list(A); +to_list(B) when is_binary(B) -> unicode:characters_to_list(B); +to_list(I) when is_integer(I) -> integer_to_list(I); +to_list(L) when is_list(L) -> L. diff --git a/src/py_isolated.erl b/src/py_isolated.erl index 566c4f8..39c3a92 100644 --- a/src/py_isolated.erl +++ b/src/py_isolated.erl @@ -81,7 +81,6 @@ -define(DEFAULT_RESTART_PERIOD_MS, 10000). -define(SHUTDOWN_GRACE_MS, 1000). -define(EXIT_STATUS_WAIT_MS, 5000). --define(SOCKET_BUF, 1024 * 1024). -record(child, { port :: port(), @@ -198,7 +197,7 @@ handle_event(info, {Port, {data, Out}}, _State, #data{child = #child{port = Port log_output(Data, Out), keep_state_and_data; handle_event(info, {Port, {exit_status, Status}}, State, #data{child = #child{port = Port}} = Data) -> - child_exited(exit_reason(Status), State, Data); + child_exited(py_child:exit_reason(Status), State, Data); handle_event(info, {Port, _}, _State, _Data) when is_port(Port) -> keep_state_and_data; handle_event(state_timeout, exit_status, {restarting, _} = State, #data{child = Child} = Data) -> @@ -401,44 +400,11 @@ format_status(#{data := #data{child = #child{} = Child} = Data} = Status) -> format_status(Status) -> Status. -%% @doc Python executable used for isolated children: the `python' option, -%% then the `isolated_python' application env, then the interpreter matching -%% the embedded runtime, then `python3' from PATH. +%% @doc Python executable used for isolated children; see +%% py_child:python_executable/1. -spec python_executable(map()) -> string() | {error, term()}. python_executable(Opts) -> - Candidate = case maps:get(python, Opts, undefined) of - undefined -> - case application:get_env(erlang_python, isolated_python) of - {ok, P} -> P; - undefined -> default_python() - end; - P -> P - end, - resolve_exe(to_list(Candidate)). - -default_python() -> - case persistent_term:get({?MODULE, python}, undefined) of - undefined -> - Exe = try py:python_executable() catch _:_ -> "python3" end, - persistent_term:put({?MODULE, python}, Exe), - Exe; - Exe -> - Exe - end. - -resolve_exe(Exe) -> - case filename:pathtype(Exe) of - absolute -> - case filelib:is_file(Exe) of - true -> Exe; - false -> {error, {python_not_found, Exe}} - end; - _ -> - case os:find_executable(Exe) of - false -> {error, {python_not_found, Exe}}; - Found -> Found - end - end. + py_child:python_executable(Opts). %% ============================================================================ %% Child startup @@ -474,28 +440,24 @@ start_child_1(#data{opts = Opts} = St) -> end. spawn_child(Python, Opts) -> - Dir = sock_dir(), - Path = filename:join(Dir, "ctx_" ++ integer_to_list(erlang:unique_integer([positive])) ++ ".sock"), - _ = file:delete(Path), - case socket:open(local, stream, default) of + Path = py_child:new_sock_path("ctx_"), + case py_child:listen(Path) of {ok, L} -> try - ok = socket:bind(L, #{family => local, path => Path}), - ok = socket:listen(L), - Script = filename:join(priv_dir(), "py_isolated_child.py"), - Args = [Script, Path | rlimit_args(Opts) ++ cgroup_args(Opts)], + Script = filename:join(py_child:priv_dir(), "py_isolated_child.py"), + Args = [Script, Path | py_child:rlimit_args(Opts) ++ py_child:cgroup_args(Opts)], PortOpts = [exit_status, stderr_to_stdout, binary, use_stdio, - {args, Args}, {env, env_opt(Opts)}], + {args, Args}, {env, py_child:port_env(Opts)}], Port = open_port({spawn_executable, Python}, PortOpts), OsPid = case erlang:port_info(Port, os_pid) of {os_pid, Pid} -> Pid; _ -> 0 end, Timeout = maps:get(start_timeout, Opts, ?DEFAULT_START_TIMEOUT_MS), - case accept_child(L, Port, Timeout) of + case py_child:accept(L, {port, Port}, Timeout) of {ok, S} -> _ = file:delete(Path), - tune_socket(S), + py_child:tune_socket(S), {ok, #child{port = Port, os_pid = OsPid, listener = L, sock = S, sock_path = Path}}; {error, Reason} -> @@ -510,56 +472,10 @@ spawn_child(Python, Opts) -> socket:close(L), {error, {spawn_failed, {Class, Err, Stack}}} end; + {error, {socket_open_failed, _}} = Err -> + Err; {error, Reason} -> - {error, {socket_open_failed, Reason}} - end. - -%% Accept while also watching the port: a child that dies before connecting -%% (bad interpreter, missing script) is reported with its output. -accept_child(L, Port, Timeout) -> - Deadline = erlang:monotonic_time(millisecond) + Timeout, - accept_child(L, Port, Deadline, []). - -accept_child(L, Port, Deadline, Out) -> - case socket:accept(L, nowait) of - {ok, S} -> - %% Output printed before connecting is still worth logging - [self() ! {Port, {data, D}} || D <- lists:reverse(Out)], - {ok, S}; - {select, {select_info, _, Handle}} -> - Left = max(0, Deadline - erlang:monotonic_time(millisecond)), - receive - {'$socket', L, select, Handle} -> - accept_child(L, Port, Deadline, Out); - {Port, {exit_status, Status}} -> - _ = socket:cancel(L, {select_info, accept, Handle}), - {error, {child_exited_at_start, exit_reason(Status), - drain_port_output(Port, Out)}}; - {Port, {data, D}} -> - %% Keep it here, not in the mailbox: re-sending it would - %% make this receive return at once and never time out - accept_child(L, Port, Deadline, [D | Out]) - after Left -> - _ = socket:cancel(L, {select_info, accept, Handle}), - {error, {start_timeout, drain_port_output(Port, Out)}} - end; - {error, Reason} -> - {error, {accept_failed, Reason}} - end. - -%% Default Unix socket buffers are small (8 KB on macOS); large payloads -%% would cross in hundreds of wakeups. Best effort: the kernel clamps. -tune_socket(S) -> - _ = socket:setopt(S, {otp, rcvbuf}, ?SOCKET_BUF), - _ = socket:setopt(S, {socket, rcvbuf}, ?SOCKET_BUF), - _ = socket:setopt(S, {socket, sndbuf}, ?SOCKET_BUF), - ok. - -drain_port_output(Port, Acc) -> - receive - {Port, {data, D}} -> drain_port_output(Port, [D | Acc]) - after 50 -> - iolist_to_binary(lists:reverse(Acc)) + {error, {spawn_failed, Reason}} end. %% Blocking handshake: ready event, init request, then the preload exec. @@ -569,10 +485,10 @@ handshake(#data{child = Child, opts = Opts} = St0) -> case recv_frame_sync(Child, Timeout) of {ok, {0, ?STATUS_EVENT, {ready, Info}}, Child1} -> St1 = St#data{child = Child1#child{info = Info}}, - Paths = [to_bin(P) || P <- py_import:all_paths()] ++ extra_paths(Opts), + Paths = [py_child:to_bin(P) || P <- py_import:all_paths()] ++ extra_paths(Opts), %% Registered imports are pre-cached in sys.modules, as %% interp_apply_imports does for the embedded modes - Imports = lists:usort([to_bin(M) || {M, _} <- py_import:all_imports()]), + Imports = lists:usort([py_child:to_bin(M) || {M, _} <- py_import:all_imports()]), case sync_request(St1, {init, self(), Paths, Imports}, Timeout) of {{ok, _}, St2} -> run_preload(St2, Timeout); @@ -624,7 +540,7 @@ py_preload_code() -> end. extra_paths(Opts) -> - [to_bin(P) || P <- maps:get(paths, Opts, [])]. + [py_child:to_bin(P) || P <- maps:get(paths, Opts, [])]. %% Send a request and wait for its reply, ignoring nothing: callbacks made %% by the child during startup are served too. @@ -673,7 +589,7 @@ sync_wait(#data{child = Child} = St, Id, Timeout) -> end. recv_frame_sync(#child{buf = Buf} = Child, Timeout) -> - case parse_frame(Buf) of + case py_child:parse_frame(Buf) of {ok, Frame, Rest} -> {ok, Frame, Child#child{buf = Rest}}; more -> @@ -739,7 +655,7 @@ pass_fd(_From, _MRef, _Fd, {restarting, _}, _Data) -> {keep_state_and_data, [postpone]}; pass_fd(From, MRef, Fd, _State, #data{child = Child, next_id = Id, pending = Pending} = Data) when is_integer(Fd), Fd >= 0 -> - Frame = frame(Id, ?STATUS_REQUEST, term_to_binary(pass_fd)), + Frame = py_child:frame(Id, ?STATUS_REQUEST, term_to_binary(pass_fd)), Msg = #{iov => [Frame], ctrl => [#{level => socket, type => rights, data => <>}]}, case socket:sendmsg(Child#child.sock, Msg) of @@ -789,7 +705,7 @@ drain_socket(State, #data{child = #child{sock = S, buf = Buf} = Child} = Data, A process_frames({restarting, _} = State, Data, Actions) -> {State, Data, Actions}; process_frames(State, #data{child = #child{buf = Buf} = Child} = Data, Actions) -> - case parse_frame(Buf) of + case py_child:parse_frame(Buf) of {ok, Frame, Rest} -> Data1 = Data#data{child = Child#child{buf = Rest}}, {State1, Data2, Actions1} = handle_frame(Frame, State, Data1, Actions), @@ -800,24 +716,6 @@ process_frames(State, #data{child = #child{buf = Buf} = Child} = Data, Actions) socket_broken({malformed_frame, Reason}, State, Data, Actions) end. -parse_frame(<>) -> - case Body of - <> -> - try - Term = case Payload of - <<>> -> undefined; - _ -> binary_to_term(Payload) - end, - {ok, {Id, Status, Term}, Rest} - catch - error:badarg -> {error, bad_etf} - end; - <<>> -> - {error, empty_body} - end; -parse_frame(_) -> - more. - handle_frame({Id, Status, Term}, State, #data{pending = Pending} = Data, Actions) when Status =:= ?STATUS_OK; Status =:= ?STATUS_ERROR -> @@ -904,7 +802,7 @@ run_callback({call, Name, Args}) -> T when is_tuple(T) -> tuple_to_list(T); _ -> [Args] end, - try py_callback:execute(to_bin(Name), ArgsList) of + try py_callback:execute(py_child:to_bin(Name), ArgsList) of {ok, Result} -> {?STATUS_OK, Result}; {error, {not_found, N}} -> @@ -980,8 +878,8 @@ kill(_Reason, _State, _Data) -> keep_state_and_data. kill_port(Port, OsPid) -> - case OsPid > 0 andalso erlang:port_info(Port) =/= undefined of - true -> _ = py_nif:os_kill(OsPid, 9), ok; + case erlang:port_info(Port) =/= undefined of + true -> py_child:kill_os_pid(OsPid); false -> ok end. @@ -1008,9 +906,6 @@ enter_restarting(Reason, _State, Data, Actions) -> {{restarting, Reason}, Data#data{kill_target = undefined}, [{{timeout, kill}, cancel} | Actions]}. -exit_reason(Status) when Status > 128 -> {signal, Status - 128}; -exit_reason(Status) -> {exit_status, Status}. - %% The port reported the child's exit: fail what was in flight, then %% restart within the budget or stop. child_exited(Reason, State, #data{child = Child, opts = Opts} = Data0) -> @@ -1134,16 +1029,8 @@ loop_exited(_Result, Data) -> %% Wire helpers %% --------------------------------------------------------------------------- -frame(Id, Status, Payload) -> - Body = <>, - <>. - send_frame(#child{sock = S}, Id, Status, Term) -> - case socket:send(S, frame(Id, Status, term_to_binary(Term))) of - ok -> ok; - {error, {Reason, _Rest}} -> {error, Reason}; - {error, Reason} -> {error, Reason} - end. + py_child:send_frame(S, Id, Status, Term). log_output(#data{id = Id, child = #child{os_pid = OsPid}}, Data) -> Lines = binary:split(Data, <<"\n">>, [global, trim_all]), @@ -1155,45 +1042,3 @@ log_event(#data{id = Id}, Level, Msg) -> error -> error; warning -> warning; debug -> debug; _ -> info end, logger:log(Lvl, "py_context ~p (isolated): ~s", [Id, Msg]). - -sock_dir() -> - Base = case os:getenv("TMPDIR") of - false -> "/tmp"; - T -> T - end, - Dir = filename:join(Base, "erlang_python_" ++ os:getpid()), - ok = filelib:ensure_dir(filename:join(Dir, "x")), - _ = file:change_mode(Dir, 8#700), - Dir. - -priv_dir() -> - case code:priv_dir(erlang_python) of - {error, bad_name} -> - filename:join(filename:dirname(filename:dirname(code:which(?MODULE))), "priv"); - Dir -> - Dir - end. - -rlimit_args(Opts) -> - Limits = maps:get(rlimits, Opts, #{}), - lists:append([case maps:get(K, Limits, undefined) of - undefined -> []; - V when is_integer(V), V >= 0 -> ["--rlimit-" ++ atom_to_list(K), integer_to_list(V)] - end || K <- [as, cpu, nofile]]). - -cgroup_args(Opts) -> - case maps:get(cgroup, Opts, undefined) of - undefined -> []; - Dir -> ["--cgroup", to_list(Dir)] - end. - -env_opt(Opts) -> - [{to_list(K), to_list(V)} || {K, V} <- maps:to_list(maps:get(env, Opts, #{}))]. - -to_bin(A) when is_atom(A) -> atom_to_binary(A, utf8); -to_bin(L) when is_list(L) -> unicode:characters_to_binary(L); -to_bin(B) when is_binary(B) -> B. - -to_list(A) when is_atom(A) -> atom_to_list(A); -to_list(B) when is_binary(B) -> unicode:characters_to_list(B); -to_list(L) when is_list(L) -> L. From 4a8da03b582fd2c0c1d9364d8178a645c8bbe084 Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Sat, 26 Sep 2026 10:32:31 +0200 Subject: [PATCH 02/18] Let an isolated child start without the VM's environment clear_env => true gives the child only the variables named in env, and hash_seed pins PYTHONHASHSEED, so two children built the same way hash strings and order sets the same way. Bad values are refused when the context starts. --- docs/isolated.md | 2 ++ src/py_child.erl | 39 ++++++++++++++++++++++++++--- src/py_isolated.erl | 9 +++++-- test/py_isolated_SUITE.erl | 51 ++++++++++++++++++++++++++++++++++++++ 4 files changed, 96 insertions(+), 5 deletions(-) diff --git a/docs/isolated.md b/docs/isolated.md index 9705a70..067be8c 100644 --- a/docs/isolated.md +++ b/docs/isolated.md @@ -50,6 +50,8 @@ Options of `py_context:new/1` specific to this mode: | `rlimits` | `#{}` | `#{as => Bytes, cpu => Seconds, nofile => N}`, applied with `setrlimit` before any user code | | `cgroup` | none | Path of a cgroup v2 directory the child joins (limits written by you: `memory.max`, `cpu.max`, `pids.max`) | | `env` | `#{}` | Extra environment variables for the child | +| `clear_env` | `false` | When `true` the child inherits nothing from the VM's environment: it sees only `env` | +| `hash_seed` | `random` | `PYTHONHASHSEED` for the child (0 to 4294967295): the same seed gives the same `set` and `dict`-of-`str` iteration order in every child | | `paths` | `[]` | Extra `sys.path` entries (registered `py_import` paths and imports are applied too) | | `preload` | none | Code run once in the child before anything else | | `kill_after` | `1000` | Milliseconds between a soft interrupt and `SIGKILL` | diff --git a/src/py_child.erl b/src/py_child.erl index 47add4b..b2f4793 100644 --- a/src/py_child.erl +++ b/src/py_child.erl @@ -33,6 +33,7 @@ accept/3, tune_socket/1, port_env/1, + check_env_opts/1, rlimit_args/1, cgroup_args/1, frame/3, @@ -184,11 +185,43 @@ tune_socket(S) -> _ = socket:setopt(S, {socket, sndbuf}, ?SOCKET_BUF), ok. -%% @doc The `{env, ...}' port option for a child: `env' adds variables to -%% what the child inherits from the VM. +%% @doc The `{env, ...}' port option for a child. +%% +%% `env' adds variables to what the child inherits from the VM. With +%% `clear_env => true' nothing is inherited: every variable of the VM not +%% named in `env' is unset. `hash_seed' sets PYTHONHASHSEED; `random' (the +%% default) leaves the choice to Python. -spec port_env(map()) -> [{string(), string() | false}]. port_env(Opts) -> - [{to_list(K), to_list(V)} || {K, V} <- maps:to_list(maps:get(env, Opts, #{}))]. + Env = [{to_list(K), to_list(V)} || {K, V} <- maps:to_list(maps:get(env, Opts, #{}))], + Seed = case maps:get(hash_seed, Opts, random) of + random -> []; + N -> [{"PYTHONHASHSEED", integer_to_list(N)}] + end, + Set = Env ++ Seed, + Clear = case maps:get(clear_env, Opts, false) of + true -> + Keep = [K || {K, _} <- Set], + lists:usort([{K, false} || KV <- os:getenv(), + K <- [hd(string:split(KV, "="))], + K =/= "", not lists:member(K, Keep)]); + false -> + [] + end, + Clear ++ Set. + +%% @doc Check the options port_env/1 reads. +-spec check_env_opts(map()) -> ok | {error, term()}. +check_env_opts(Opts) -> + case {maps:get(hash_seed, Opts, random), maps:get(clear_env, Opts, false)} of + {Seed, _} when Seed =/= random, + not (is_integer(Seed) andalso Seed >= 0 andalso Seed =< 4294967295) -> + {error, {badarg, {hash_seed, Seed}}}; + {_, Clear} when not is_boolean(Clear) -> + {error, {badarg, {clear_env, Clear}}}; + _ -> + ok + end. -spec rlimit_args(map()) -> [string()]. rlimit_args(Opts) -> diff --git a/src/py_isolated.erl b/src/py_isolated.erl index 39c3a92..1109597 100644 --- a/src/py_isolated.erl +++ b/src/py_isolated.erl @@ -412,8 +412,13 @@ python_executable(Opts) -> start_child(#data{opts = Opts} = St) -> case check_platform_opts(Opts) of - ok -> start_child_1(St); - {error, _} = Err -> Err + ok -> + case py_child:check_env_opts(Opts) of + ok -> start_child_1(St); + {error, _} = Err -> Err + end; + {error, _} = Err -> + Err end. %% cgroups exist only on Linux; rlimits are POSIX and apply everywhere. diff --git a/test/py_isolated_SUITE.erl b/test/py_isolated_SUITE.erl index 937bd9d..0b6ba29 100644 --- a/test/py_isolated_SUITE.erl +++ b/test/py_isolated_SUITE.erl @@ -64,6 +64,9 @@ test_startup_error_reported/1, test_cgroup_option_platform/1, test_env_option/1, + test_clear_env_option/1, + test_hash_seed_option/1, + test_bad_env_options/1, test_preload_option/1 ]). @@ -119,6 +122,9 @@ groups() -> test_startup_error_reported, test_cgroup_option_platform, test_env_option, + test_clear_env_option, + test_hash_seed_option, + test_bad_env_options, test_preload_option ], [{worker, [], RoundTrip}, @@ -786,6 +792,51 @@ test_env_option(Config) -> {ok, <<"yes">>} = py_context:eval(C, <<"__import__('os').environ.get('PY_ISOLATED_PROBE')">>), stop(C). +%% The VM's environment reaches the child unless clear_env is set; with it +%% the child sees only what `env' names. +test_clear_env_option(Config) -> + true = os:putenv("PY_ISOLATED_LEAK", "from-the-vm"), + try + Probe = <<"(os.environ.get('PY_ISOLATED_LEAK'), os.environ.get('PY_ISOLATED_KEEP'))">>, + Inherit = new_ctx(Config, #{env => #{"PY_ISOLATED_KEEP" => "1"}}), + ok = py_context:exec(Inherit, <<"import os">>), + {ok, {<<"from-the-vm">>, <<"1">>}} = py_context:eval(Inherit, Probe), + stop(Inherit), + Clean = new_ctx(Config, #{clear_env => true, env => #{"PY_ISOLATED_KEEP" => "1"}}), + ok = py_context:exec(Clean, <<"import os">>), + {ok, {none, <<"1">>}} = py_context:eval(Clean, Probe), + {ok, Names} = py_context:eval(Clean, <<"sorted(os.environ)">>), + %% Python itself may add a few (e.g. __CF_USER_TEXT_ENCODING on macOS) + false = lists:member(<<"HOME">>, Names), + false = lists:member(<<"PATH">>, Names), + stop(Clean) + after + os:unsetenv("PY_ISOLATED_LEAK") + end. + +%% Same seed, same str hashes and set order in two separate children; another +%% seed changes them. +test_hash_seed_option(Config) -> + Probe = <<"(hash('erlang-python'), repr(list({'a','b','c','d','e','f','g','h'})))">>, + Run = fun(Seed) -> + C = new_ctx(Config, #{hash_seed => Seed}), + {ok, R} = py_context:eval(C, Probe), + {ok, Env} = py_context:eval(C, <<"__import__('os').environ.get('PYTHONHASHSEED')">>), + stop(C), + {R, Env} + end, + {A, <<"7">>} = Run(7), + {A, <<"7">>} = Run(7), + {B, <<"8">>} = Run(8), + true = A =/= B, + ok. + +test_bad_env_options(_Config) -> + {error, {badarg, {hash_seed, -1}}} = py_context:new(#{mode => isolated, hash_seed => -1}), + {error, {badarg, {hash_seed, 1 bsl 32}}} = py_context:new(#{mode => isolated, hash_seed => 1 bsl 32}), + {error, {badarg, {clear_env, yes}}} = py_context:new(#{mode => isolated, clear_env => yes}), + ok. + test_preload_option(Config) -> C = new_ctx(Config, #{preload => <<"preloaded = 'yes'">>}), {ok, <<"yes">>} = py_context:eval(C, <<"preloaded">>), From 74df43807f31c87540e0107cc128531a785112a6 Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Sat, 26 Sep 2026 10:42:00 +0200 Subject: [PATCH 03/18] Let py_isolated drive a session child A context started with session => true runs in its own scratch directory, is killed rather than asked to stop, and is never restarted: once its child is gone it answers every request with the exit reason until it is closed. Its child can be forked by a session template instead of spawned; the template reports the exit by message. The child runtime can be built before it is connected, and the serve loop is shared by both kinds of child. --- priv/_erlang_impl/_isolated.py | 12 +- priv/py_isolated_child.py | 17 ++- src/py_child.erl | 40 +++++-- src/py_isolated.erl | 206 ++++++++++++++++++++++++++++----- 4 files changed, 234 insertions(+), 41 deletions(-) diff --git a/priv/_erlang_impl/_isolated.py b/priv/_erlang_impl/_isolated.py index 04f6a97..c17b5a1 100644 --- a/priv/_erlang_impl/_isolated.py +++ b/priv/_erlang_impl/_isolated.py @@ -142,7 +142,8 @@ def _callback_error(reason): class Runtime: - """One per child process.""" + """One per child process. A session template's zygote builds it with + sock=None; each forked session sets sock before start().""" def __init__(self, sock, context_pid=None): self.sock = sock @@ -198,7 +199,14 @@ def _signal_main(self): # -- writing ----------------------------------------------------------- + def _check_connected(self): + if self.sock is None: + # Built by a session template's zygote, not connected yet + raise RuntimeError('erlang is not connected: this code runs while a ' + 'session template is prepared, before any session') + def _write_frame(self, frame_id, status, payload): + self._check_connected() body = bytes([status]) + payload data = _HEADER.pack(frame_id, len(body)) + body # A signal landing inside sendall would tear the frame and @@ -242,6 +250,7 @@ def request(self, term, timeout=None): On the main thread the wait also serves requests coming from Erlang, so nested calls work. Returns (status, value).""" + self._check_connected() if self.broken: raise PipeBroken(self.broken_reason) on_main = threading.current_thread() is self.main_thread @@ -265,6 +274,7 @@ def request(self, term, timeout=None): def request_async(self, term): """Send a status-3 request; returns an asyncio Future for the reply.""" + self._check_connected() if self.broken: raise PipeBroken(self.broken_reason) loop = asyncio.get_running_loop() diff --git a/priv/py_isolated_child.py b/priv/py_isolated_child.py index 006e074..6c0b393 100644 --- a/priv/py_isolated_child.py +++ b/priv/py_isolated_child.py @@ -183,13 +183,21 @@ def main(argv): _die('cannot connect to %s: %s' % (opts['socket'], exc)) from _erlang_impl import _isolated - from _erlang_impl._etf import Atom runtime = _isolated.Runtime(sock) _isolated.install_erlang_module(runtime) + serve(runtime, opts['rlimits'], rlimit_errors, cgroup_error) + + +def serve(runtime, limits, rlimit_errors, cgroup_error): + """Start a connected runtime, report ready (or the start-up problems) and + serve Erlang until it closes the socket. Shared by the spawned child and + the sessions forked by py_zygote.py. Never returns.""" + from _erlang_impl._etf import Atom + runtime.start() - if _AS_VIA_WATCHDOG and 'as' in opts['rlimits']: - _start_memory_watchdog(opts['rlimits']['as'], runtime) + if _AS_VIA_WATCHDOG and 'as' in limits: + _start_memory_watchdog(limits['as'], runtime) if rlimit_errors or cgroup_error: problems = [(Atom('rlimit'), Atom(k), msg) for k, msg in rlimit_errors] @@ -212,11 +220,10 @@ def main(argv): runtime.serve_forever() finally: try: - sock.close() + runtime.sock.close() except OSError: pass os._exit(0) - if __name__ == '__main__': main(sys.argv) diff --git a/src/py_child.erl b/src/py_child.erl index b2f4793..db22e81 100644 --- a/src/py_child.erl +++ b/src/py_child.erl @@ -40,6 +40,8 @@ send_frame/4, parse_frame/1, exit_reason/1, + code_reason/1, + os_pid_alive/1, kill_os_pid/1, to_bin/1, to_list/1]). @@ -131,10 +133,12 @@ listen(Path) -> {error, {socket_open_failed, Reason}} end. -%% @doc Accept the child's connection while watching its port, so a child -%% that dies before connecting (bad interpreter, missing script) is reported -%% with its output instead of timing out. --spec accept(socket:socket(), {port, port()}, timeout()) -> +%% @doc Accept the child's connection while watching for its death, so a +%% child that dies before connecting (bad interpreter, missing script) is +%% reported instead of timing out. `{port, Port}' watches a spawned child and +%% keeps its output for the error; `{os_pid, Pid}' watches a forked one +%% through the `{py_session_exited, Pid, Code}' message its template sends. +-spec accept(socket:socket(), {port, port()} | {os_pid, pos_integer()}, timeout()) -> {ok, socket:socket()} | {error, term()}. accept(L, Watch, Timeout) -> Deadline = erlang:monotonic_time(millisecond) + Timeout, @@ -144,8 +148,7 @@ accept(L, Watch, Deadline, Out) -> case socket:accept(L, nowait) of {ok, S} -> %% Output printed before connecting is still worth logging - {port, Port} = Watch, - [self() ! {Port, {data, D}} || D <- lists:reverse(Out)], + [self() ! {Port, {data, D}} || {port, Port} <- [Watch], D <- lists:reverse(Out)], {ok, S}; {select, {select_info, _, Handle}} -> Left = max(0, Deadline - erlang:monotonic_time(millisecond)), @@ -159,11 +162,17 @@ accept(L, Watch, Deadline, Out) -> {Port, {data, D}} when Watch =:= {port, Port} -> %% Keep it here, not in the mailbox: re-sending it would %% make this receive return at once and never time out - accept(L, Watch, Deadline, [D | Out]) + accept(L, Watch, Deadline, [D | Out]); + {py_session_exited, Pid, Code} when Watch =:= {os_pid, Pid} -> + _ = socket:cancel(L, {select_info, accept, Handle}), + {error, {child_exited_at_start, code_reason(Code), <<>>}} after Left -> - {port, Port} = Watch, _ = socket:cancel(L, {select_info, accept, Handle}), - {error, {start_timeout, drain_port_output(Port, Out)}} + Output = case Watch of + {port, Port} -> drain_port_output(Port, Out); + _ -> <<>> + end, + {error, {start_timeout, Output}} end; {error, Reason} -> {error, {accept_failed, Reason}} @@ -276,6 +285,12 @@ parse_frame(_) -> exit_reason(Status) when Status > 128 -> {signal, Status - 128}; exit_reason(Status) -> {exit_status, Status}. +%% @doc Exit reason from an `os.waitstatus_to_exitcode' value (-N is +%% signal N), as the zygote reports it. +-spec code_reason(integer()) -> {signal, pos_integer()} | {exit_status, integer()}. +code_reason(Code) when Code < 0 -> {signal, -Code}; +code_reason(Code) -> {exit_status, Code}. + -spec kill_os_pid(integer()) -> ok. kill_os_pid(OsPid) when is_integer(OsPid), OsPid > 0 -> _ = py_nif:os_kill(OsPid, 9), @@ -283,6 +298,13 @@ kill_os_pid(OsPid) when is_integer(OsPid), OsPid > 0 -> kill_os_pid(_) -> ok. +%% @doc Whether a process still exists (a zombie counts until reaped). +-spec os_pid_alive(integer()) -> boolean(). +os_pid_alive(OsPid) when is_integer(OsPid), OsPid > 0 -> + py_nif:os_kill(OsPid, 0) =/= {error, esrch}; +os_pid_alive(_) -> + false. + to_bin(A) when is_atom(A) -> atom_to_binary(A, utf8); to_bin(L) when is_list(L) -> unicode:characters_to_binary(L); to_bin(B) when is_binary(B) -> B. diff --git a/src/py_isolated.erl b/src/py_isolated.erl index 1109597..70ca250 100644 --- a/src/py_isolated.erl +++ b/src/py_isolated.erl @@ -44,8 +44,20 @@ %%%
  • `{restarting, Reason}' - the child is gone or being killed; %%% requests are postponed until the new child is up. In-flight %%% requests fail with `{error, Reason}'.
  • +%%%
  • `{exited, Reason}' - a session's child is gone. Sessions are never +%%% restarted: every request answers `{error, Reason}' until the +%%% context is stopped.
  • %%% %%% +%%% == Sessions == +%%% +%%% `py_session' starts contexts with `session => true' and an `origin'. +%%% `{fork, Template}' asks a `py_session_template' to fork the child from +%%% its zygote: there is no port, the template reports the exit with +%%% `{py_session_exited, OsPid, Code}', and the child already holds the +%%% template's imports and preload. `spawn' starts one as usual. Either way +%%% the child runs in a fresh scratch directory removed when it stops. +%%% %%% The message protocol with py_context is unchanged: requests are plain %%% messages `{call, From, MRef, ...}' answered with `From ! {MRef, Reply}'. %%% Use `sys:get_state/1' to see the state and `sys:trace/2' for events. @@ -81,9 +93,11 @@ -define(DEFAULT_RESTART_PERIOD_MS, 10000). -define(SHUTDOWN_GRACE_MS, 1000). -define(EXIT_STATUS_WAIT_MS, 5000). +-define(EXIT_PROBE_MS, 100). -record(child, { - port :: port(), + %% undefined for a child forked by a session template + port :: port() | undefined, os_pid :: pos_integer(), listener :: socket:socket() | undefined, sock :: socket:socket(), @@ -118,14 +132,16 @@ %% Request id (or `loop') the armed kill backstop is bound to kill_target :: pos_integer() | loop | undefined, %% Callers of kill/1 answered once the new child is up - kill_waiters = [] :: [{pid(), reference()}] + kill_waiters = [] :: [{pid(), reference()}], + %% Session scratch directory (the child's cwd), removed on stop + scratch :: file:filename() | undefined }). -define(IS_MAIN(K), (K =:= call orelse K =:= eval orelse K =:= exec orelse (is_tuple(K) andalso element(1, K) =:= start_loop))). -type state() :: idle | {busy, pos_integer()} | looping | stopping_loop - | {restarting, term()}. + | {restarting, term()} | {exited, term()}. -export_type([state/0]). %% ============================================================================ @@ -176,6 +192,10 @@ init(_Args) -> handle_event(enter, _Old, idle, #data{kill_waiters = Waiters} = Data) -> [W ! {M, ok} || {W, M} <- Waiters], {keep_state, Data#data{kill_waiters = []}}; +handle_event(enter, _Old, {restarting, _}, #data{child = #child{port = undefined}}) -> + %% A forked child's exit comes from its template, which may be gone: + %% check the pid too + {keep_state_and_data, [{state_timeout, ?EXIT_PROBE_MS, {probe_exit, 0}}]}; handle_event(enter, _Old, {restarting, _}, _Data) -> %% SIGKILL was sent (or the child is exiting): the port reports it %% within milliseconds; this is the safety net @@ -200,6 +220,22 @@ handle_event(info, {Port, {exit_status, Status}}, State, #data{child = #child{po child_exited(py_child:exit_reason(Status), State, Data); handle_event(info, {Port, _}, _State, _Data) when is_port(Port) -> keep_state_and_data; +handle_event(info, {py_session_exited, OsPid, Code}, State, + #data{child = #child{port = undefined, os_pid = OsPid}} = Data) -> + child_exited(py_child:code_reason(Code), State, Data); +handle_event(info, {py_session_exited, _, _}, _State, _Data) -> + keep_state_and_data; +handle_event(state_timeout, {probe_exit, Waited}, {restarting, _} = State, + #data{child = #child{os_pid = OsPid}} = Data) -> + case py_child:os_pid_alive(OsPid) of + false -> + child_exited({signal, 9}, State, Data); + true when Waited >= ?EXIT_STATUS_WAIT_MS -> + handle_event(state_timeout, exit_status, State, Data); + true -> + {keep_state_and_data, + [{state_timeout, ?EXIT_PROBE_MS, {probe_exit, Waited + ?EXIT_PROBE_MS}}]} + end; handle_event(state_timeout, exit_status, {restarting, _} = State, #data{child = Child} = Data) -> logger:error("py_context ~p (isolated): child ~p did not exit after SIGKILL", [Data#data.id, Child#child.os_pid]), @@ -298,6 +334,9 @@ handle_event({timeout, kill}, Target, State, #data{pending = Pending} = Data) -> false -> {keep_state, Data#data{kill_target = undefined}} end; +handle_event(info, {kill, From, MRef}, {exited, _}, _Data) -> + From ! {MRef, ok}, + keep_state_and_data; handle_event(info, {kill, From, MRef}, State, Data) -> case State of {restarting, _} -> @@ -306,11 +345,26 @@ handle_event(info, {kill, From, MRef}, State, Data) -> _ -> kill(killed, State, Data#data{kill_waiters = [{From, MRef} | Data#data.kill_waiters]}) end; -handle_event(info, {stop, From, MRef}, _State, Data) -> - Data1 = stop_child(Data, graceful), +handle_event(info, {stop, From, MRef}, _State, #data{opts = Opts} = Data) -> + How = case maps:get(session, Opts, false) of + true -> kill; + false -> graceful + end, + Data1 = stop_child(Data, How), From ! {MRef, ok}, {stop, normal, Data1}; +%% A session started ahead of time by its template is handed to the process +%% that asked for it: that process becomes the parent (linked, its crash +%% stops the session) and the template lets go. +handle_event(info, {set_parent, From, MRef, NewParent}, _State, #data{parent = Old} = Data) -> + link(NewParent), + _ = Old =/= NewParent andalso unlink(Old), + %% An EXIT from the old parent may already be queued: forget it + receive {'EXIT', Old, _} -> ok after 0 -> ok end, + From ! {MRef, ok}, + {keep_state, Data#data{parent = NewParent}}; + %% ---- worker loop ----------------------------------------------------------- handle_event(info, {stop_loop, From, MRef, GraceMs}, looping, #data{loop = Loop} = Data) -> @@ -385,9 +439,14 @@ handle_event(info, _Other, _State, _Data) -> terminate(Reason, _State, #data{child = Child} = Data) -> _ = case Child of undefined -> Data; - _ when Reason =:= normal; Reason =:= shutdown -> stop_child(Data, graceful); + _ when Reason =:= normal; Reason =:= shutdown -> + case maps:get(session, Data#data.opts, false) of + true -> stop_child(Data, kill); + false -> stop_child(Data, graceful) + end; _ -> stop_child(Data, kill) end, + remove_scratch(Data), ets:delete(?REF_TAB, self()), ok. @@ -431,17 +490,74 @@ check_platform_opts(Opts) -> {_, {unix, Os}} -> {error, {cgroup_unsupported, Os}} end. -start_child_1(#data{opts = Opts} = St) -> - case python_executable(Opts) of - {error, _} = Err -> - Err; - Python -> - case spawn_child(Python, Opts) of - {ok, Child} -> - handshake(St#data{child = Child}); - {error, _} = Err -> - Err +start_child_1(#data{opts = Opts} = St0) -> + St = with_scratch(St0), + ChildOpts = case St#data.scratch of + undefined -> Opts; + Dir -> Opts#{cd => Dir} + end, + Started = case maps:get(origin, Opts, spawn) of + {fork, Template} -> + fork_child(Template, ChildOpts); + spawn -> + case python_executable(Opts) of + {error, _} = NoPython -> NoPython; + Python -> spawn_child(Python, ChildOpts) end + end, + case Started of + {ok, Child} -> + handshake(St#data{child = Child}); + {error, _} = Failed -> + remove_scratch(St), + Failed + end. + +%% A session runs in its own empty directory, created here and removed +%% when the context stops. +with_scratch(#data{scratch = undefined, opts = #{session := true}} = St) -> + Dir = filename:join(py_child:sock_dir(), + "sess_" ++ integer_to_list(erlang:unique_integer([positive]))), + ok = file:make_dir(Dir), + St#data{scratch = Dir}; +with_scratch(St) -> + St. + +remove_scratch(#data{scratch = undefined}) -> + ok; +remove_scratch(#data{scratch = Dir}) -> + _ = file:del_dir_r(Dir), + ok. + +%% Ask the session template to fork a child from its zygote; the child +%% connects to our socket like a spawned one. +fork_child(Template, Opts) -> + Path = py_child:new_sock_path("sess_"), + case py_child:listen(Path) of + {ok, L} -> + Timeout = maps:get(start_timeout, Opts, ?DEFAULT_START_TIMEOUT_MS), + ForkOpts = maps:with([rlimits, cgroup, cd], Opts), + case py_session_template:fork(Template, Path, ForkOpts, Timeout) of + {ok, OsPid} -> + case py_child:accept(L, {os_pid, OsPid}, Timeout) of + {ok, S} -> + _ = file:delete(Path), + py_child:tune_socket(S), + {ok, #child{port = undefined, os_pid = OsPid, listener = L, + sock = S, sock_path = Path}}; + {error, Reason} -> + _ = file:delete(Path), + socket:close(L), + py_child:kill_os_pid(OsPid), + {error, Reason} + end; + {error, Reason} -> + _ = file:delete(Path), + socket:close(L), + {error, Reason} + end; + {error, Reason} -> + {error, {spawn_failed, Reason}} end. spawn_child(Python, Opts) -> @@ -452,7 +568,9 @@ spawn_child(Python, Opts) -> Script = filename:join(py_child:priv_dir(), "py_isolated_child.py"), Args = [Script, Path | py_child:rlimit_args(Opts) ++ py_child:cgroup_args(Opts)], PortOpts = [exit_status, stderr_to_stdout, binary, use_stdio, - {args, Args}, {env, py_child:port_env(Opts)}], + {args, Args}, {env, py_child:port_env(Opts)}] + ++ [{cd, Dir} || Dir <- [maps:get(cd, Opts, undefined)], + Dir =/= undefined], Port = open_port({spawn_executable, Python}, PortOpts), OsPid = case erlang:port_info(Port, os_pid) of {os_pid, Pid} -> Pid; @@ -488,6 +606,10 @@ handshake(#data{child = Child, opts = Opts} = St0) -> St = St0#data{}, Timeout = maps:get(start_timeout, Opts, ?DEFAULT_START_TIMEOUT_MS), case recv_frame_sync(Child, Timeout) of + {ok, {0, ?STATUS_EVENT, {ready, Info}}, Child1} when Child1#child.port =:= undefined -> + %% Forked: imports and preload ran in the zygote, and the child + %% learnt our pid from the fork request + {ok, St#data{child = Child1#child{info = Info}}}; {ok, {0, ?STATUS_EVENT, {ready, Info}}, Child1} -> St1 = St#data{child = Child1#child{info = Info}}, Paths = [py_child:to_bin(P) || P <- py_import:all_paths()] ++ extra_paths(Opts), @@ -624,6 +746,9 @@ request(From, MRef, {start_loop, _}, _Term, State, _Data) request(From, MRef, Kind, _Term, looping, _Data) when ?IS_MAIN(Kind) -> From ! {MRef, {error, loop_running}}, keep_state_and_data; +request(From, MRef, _Kind, _Term, {exited, Reason}, _Data) -> + From ! {MRef, {error, Reason}}, + keep_state_and_data; request(_From, _MRef, _Kind, _Term, {restarting, _}, _Data) -> {keep_state_and_data, [postpone]}; request(_From, _MRef, Kind, _Term, stopping_loop, _Data) when ?IS_MAIN(Kind) -> @@ -656,6 +781,9 @@ dispatch(From, MRef, Kind, Term, State, #data{child = Child, next_id = Id, pendi socket_broken(Reason, State, Data, []) end. +pass_fd(From, MRef, _Fd, {exited, Reason}, _Data) -> + From ! {MRef, {error, Reason}}, + keep_state_and_data; pass_fd(_From, _MRef, _Fd, {restarting, _}, _Data) -> {keep_state_and_data, [postpone]}; pass_fd(From, MRef, Fd, _State, #data{child = Child, next_id = Id, pending = Pending} = Data) @@ -882,6 +1010,8 @@ kill(Reason, State, #data{child = #child{port = Port, os_pid = OsPid}} = Data) - kill(_Reason, _State, _Data) -> keep_state_and_data. +kill_port(undefined, OsPid) -> + py_child:kill_os_pid(OsPid); kill_port(Port, OsPid) -> case erlang:port_info(Port) =/= undefined of true -> py_child:kill_os_pid(OsPid); @@ -934,6 +1064,12 @@ child_exited(Reason, State, #data{child = Child, opts = Opts} = Data0) -> logger:warning("py_context ~p (isolated): child exited: ~p", [Data0#data.id, FailReason]) end, + case maps:get(session, Opts, false) of + true -> {next_state, {exited, FailReason}, Data1}; + false -> restart_or_stop(Reason, Data0, Data1) + end. + +restart_or_stop(Reason, Data0, #data{opts = Opts} = Data1) -> case maps:get(restart, Opts, true) andalso restart_allowed(Data1) of true -> Now = erlang:monotonic_time(millisecond), @@ -983,30 +1119,48 @@ stop_child(#data{child = #child{port = Port, os_pid = OsPid} = Child} = Data, Ho case How of graceful -> _ = send_frame(Child, 0, ?STATUS_REQUEST, shutdown), - receive - {Port, {exit_status, _}} -> ok - after ?SHUTDOWN_GRACE_MS -> - kill_port(Port, OsPid), - wait_exit(Port) + case wait_exit(Child, ?SHUTDOWN_GRACE_MS) of + ok -> ok; + timeout -> kill_port(Port, OsPid), wait_exit(Child, 2000) end; _ -> kill_port(Port, OsPid), - wait_exit(Port) + wait_exit(Child, 2000) end, close_child(Child), Data1 = fail_pending({child_exited, stopped}, Data), Data1#data{child = undefined}. -wait_exit(Port) -> +wait_exit(#child{port = undefined, os_pid = OsPid}, Timeout) -> + wait_forked_exit(OsPid, Timeout); +wait_exit(#child{port = Port}, Timeout) -> receive {Port, {exit_status, _}} -> ok - after 2000 -> - ok + after Timeout -> + timeout + end. + +%% The template reports a forked child's exit; if it is gone, the pid is +%% enough +wait_forked_exit(OsPid, Timeout) -> + receive + {py_session_exited, OsPid, _} -> ok + after min(Timeout, ?EXIT_PROBE_MS) -> + case py_child:os_pid_alive(OsPid) of + false -> ok; + true when Timeout =< ?EXIT_PROBE_MS -> timeout; + true -> wait_forked_exit(OsPid, Timeout - ?EXIT_PROBE_MS) + end end. close_child(#child{port = Port, sock = S, listener = L}) -> _ = socket:close(S), _ = socket:close(L), + close_port(Port). + +close_port(undefined) -> + ok; +close_port(Port) -> try port_close(Port) catch error:badarg -> ok end, ok. From 46f8909223a493860ab06813dce6e29160c4ea5d Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Sat, 26 Sep 2026 10:49:35 +0200 Subject: [PATCH 04/18] Add isolated sessions started from a template py_session:template/1 prepares an interpreter once: imports, preload, environment and hash seed. py_session:new/1 then gives a fresh child process per session, forked from the template's zygote or spawned, and close/1 kills it. A session is an isolated context, so calls, callbacks back into the same session, interrupts and loops work unchanged. A forked session detaches from the zygote's stdio and logs its output through its own context, so a zygote that dies is seen and rebuilt while its sessions keep running. --- docs/code-map.md | 3 + priv/py_zygote.py | 258 ++++++++++++++++ src/erlang_python_sup.erl | 12 +- src/py_child.erl | 11 +- src/py_isolated.erl | 16 +- src/py_session.erl | 144 +++++++++ src/py_session_sup.erl | 51 +++ src/py_session_template.erl | 504 ++++++++++++++++++++++++++++++ test/py_session_SUITE.erl | 545 +++++++++++++++++++++++++++++++++ test/py_test_session.py | 133 ++++++++ test/py_test_session_thread.py | 6 + 11 files changed, 1675 insertions(+), 8 deletions(-) create mode 100644 priv/py_zygote.py create mode 100644 src/py_session.erl create mode 100644 src/py_session_sup.erl create mode 100644 src/py_session_template.erl create mode 100644 test/py_session_SUITE.erl create mode 100644 test/py_test_session.py create mode 100644 test/py_test_session_thread.py diff --git a/docs/code-map.md b/docs/code-map.md index 905d055..c8b1111 100644 --- a/docs/code-map.md +++ b/docs/code-map.md @@ -17,6 +17,8 @@ exercised by suites). Guides are in `docs/`, suites in `test/`. Start with | `py_context_embedded` | Process body for `worker` and `owngil` mode: the receive loop, callbacks (suspension and pipe), worker loops | live | architecture, state-machines | same | | `py_isolated` | `gen_statem` driving a child process over the socket; restart policy | live | isolated | `py_isolated_*_SUITE` | | `py_child` | Helpers shared by the processes that drive a Python child: executable lookup, socket listen/accept, frames, port env, rlimit flags | live | isolated | `py_isolated_*_SUITE` | +| `py_session` | Isolated sessions: a fresh child per session from a template (`template/1`, `new/1`, `close/1`, `run/4`, `refresh/1`) | live | sessions | `py_session_SUITE` | +| `py_session_template`, `py_session_sup` | A template: zygotes that fork sessions (`priv/py_zygote.py`) or warm spawned sessions; exit reports to the session contexts | live | sessions | `py_session_SUITE` | | `py_context_router` | Pools and scheduler-affinity routing | live | pools, context-affinity | `py_context_router_SUITE`, `py_pool_SUITE` | | `py_context_sup`, `py_context_init` | Supervisor of contexts; starts the default pool at boot | live | pools | (through the above) | | `py_nif` | Erlang stubs and docs for every NIF | live | api-reference | all | @@ -82,6 +84,7 @@ loop, channels and servers. | `_erlang_impl/_isolated.py` | Child runtime: socket frames, reader thread, re-entrant main loop, interrupt signal, asyncio loop, the `erlang` shim | isolated child | | `_erlang_impl/_shm.py` | `SharedMemory` and `SharedBuffer` wrappers over mmap | all | | `py_isolated_child.py` | Child launcher: rlimits, parent-death signal, cgroup join, connect | isolated child | +| `py_zygote.py` | Session template zygote: imports and preload once, forks one child per session, reports exits | session zygote | | `test_erlang_loop.py`, `test_async_task.py`, `test_channel_ref.py`, `tests/` | Python-side tests of the loop, tasks and channels | test | ## Tests (`test/`) diff --git a/priv/py_zygote.py b/priv/py_zygote.py new file mode 100644 index 0000000..d439c0e --- /dev/null +++ b/priv/py_zygote.py @@ -0,0 +1,258 @@ +# Copyright 2026 Benoit Chesneau +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +""" +Zygote of a session template: prepare an interpreter once, then fork one +fresh child per session. + +Started by src/py_session_template.erl as + + python3 py_zygote.py CONTROL_SOCKET_PATH + +The zygote stays single threaded (a fork only copies the calling thread) and +never serves Python requests itself. Frames on the control socket use the +format of the isolated child (_erlang_impl/_isolated.py): + + Erlang -> zygote {init, Paths, Imports, Preload} + {fork, SocketPath, ContextPid, #{rlimits, cgroup, cd}} + zygote -> Erlang replies to both, and the events + {ready, Info}, {exited, OsPid, ExitCode} + +The child runtime (_isolated.Runtime and the `erlang` module) is built here, +before the first fork, without a socket: user code can `import erlang` at +top level, and preload globals are the `__main__` namespace of every +session. After the fork the child connects the runtime to its own socket and +continues exactly like a spawned isolated child. +""" + +import os +import sys + + +def main(argv): + if len(argv) != 2: + sys.stderr.write('usage: py_zygote.py CONTROL_SOCKET_PATH\n') + os._exit(3) + priv = os.path.dirname(os.path.abspath(__file__)) + if priv not in sys.path: + sys.path.insert(0, priv) + + # Everything the session start-up path uses is imported now, once: + # imported after the fork it would be paid by every session. + import ctypes + import io + import resource # noqa: F401 + import selectors + import signal + import socket + import struct + import threading + import traceback + import py_isolated_child as child + from _erlang_impl import _etf, _isolated + from _erlang_impl._etf import Atom + ctypes.CDLL(None) + + child._arm_parent_death() + ctrl = child._connect(argv[1]) + runtime = _isolated.Runtime(None) + _isolated.install_erlang_module(runtime) + + header = struct.Struct('=QI') + + def send(frame_id, status, term): + body = bytes([status]) + _etf.encode(term) + ctrl.sendall(header.pack(frame_id, len(body)) + body) + + def event(term): + send(0, _isolated.STATUS_EVENT, term) + + def thread_names(): + return [t.name for t in threading.enumerate() if t is not threading.main_thread()] + + def init(term): + _, paths, imports, preload = term + for path in reversed(_isolated._as_list(paths)): + path = _isolated._as_text(path) + if path not in sys.path: + sys.path.insert(0, path) + import importlib + for name in _isolated._as_list(imports): + importlib.import_module(_isolated._as_text(name)) + code = _isolated._as_text(preload) + if code: + exec(compile(code, '', 'exec'), runtime.globals) + names = thread_names() + if names: + # A fork copies only this thread; the others' locks would be + # held forever in every session + return _isolated.STATUS_ERROR, (Atom('threads'), names) + return _isolated.STATUS_OK, Atom('ok') + + # SIGCHLD wakes the select loop through this pipe + rpipe, wpipe = os.pipe() + os.set_blocking(wpipe, False) + os.set_blocking(rpipe, False) + signal.set_wakeup_fd(wpipe) + signal.signal(signal.SIGCHLD, lambda *_: None) + + class LogStream(io.TextIOBase): + """sys.stdout / sys.stderr of a session: each line is logged by + the session's context in Erlang.""" + + def __init__(self, level): + self._level = Atom(level) + self._buf = '' + + def writable(self): + return True + + def write(self, text): + self._buf += text + while '\n' in self._buf: + line, self._buf = self._buf.split('\n', 1) + self._emit(line) + return len(text) + + def flush(self): + if self._buf: + line, self._buf = self._buf, '' + self._emit(line) + + def _emit(self, line): + try: + runtime.event((Atom('log'), self._level, line)) + except Exception: + pass + + def detach_stdio(runtime): + """fds 0-2 are the template's port pipes: a session holding them + would keep the port open after the zygote exits. Point them at + /dev/null and send Python-level output to Erlang instead.""" + null = os.open(os.devnull, os.O_RDWR) + for fd in (0, 1, 2): + os.dup2(null, fd) + os.close(null) + sys.stdin = open(os.devnull) + sys.stdout = LogStream('info') + sys.stderr = LogStream('warning') + + def session(path, context_pid, opts, frame_fds): + """Runs in the forked child; never returns.""" + try: + signal.set_wakeup_fd(-1) + signal.signal(signal.SIGCHLD, signal.SIG_DFL) + for fd in frame_fds: + try: + os.close(fd) + except OSError: + pass + os.setsid() + limits = {_isolated._as_text(k): v + for k, v in _isolated._as_dict(opts.get(Atom('rlimits'))).items()} + rlimit_errors = child._apply_rlimits(limits) + cgroup = opts.get(Atom('cgroup')) + cgroup_error = child._join_cgroup(_isolated._as_text(cgroup) if cgroup else None) + cd = opts.get(Atom('cd')) + if cd: + os.chdir(_isolated._as_text(cd)) + try: + sock = child._connect(_isolated._as_text(path)) + except OSError as exc: + child._die('cannot connect to %s: %s' % (path, exc)) + runtime.sock = sock + runtime.context_pid = context_pid + detach_stdio(runtime) + child.serve(runtime, limits, rlimit_errors, cgroup_error) + except BaseException: + traceback.print_exc() + finally: + os._exit(0) + + def fork(term): + _, path, context_pid, opts = term + names = thread_names() + if names: + return _isolated.STATUS_ERROR, (Atom('threads'), names) + sys.stdout.flush() + sys.stderr.flush() + pid = os.fork() + if pid == 0: + session(path, context_pid, opts if isinstance(opts, dict) else {}, + [ctrl.fileno(), rpipe, wpipe, sel_fd]) + return _isolated.STATUS_OK, pid + + def reap(): + while True: + try: + pid, status = os.waitpid(-1, os.WNOHANG) + except ChildProcessError: + return + if pid == 0: + return + event((Atom('exited'), pid, os.waitstatus_to_exitcode(status))) + + def handle(frame_id, status, term): + tag = term[0] if isinstance(term, tuple) else term + try: + if tag == 'init': + reply = init(term) + elif tag == 'fork': + reply = fork(term) + else: + reply = _isolated.STATUS_ERROR, (Atom('unknown_request'), tag) + except BaseException as exc: + reply = _isolated.STATUS_ERROR, _isolated._exc_term(exc) + send(frame_id, reply[0], reply[1]) + + info = { + Atom('os_pid'): os.getpid(), + Atom('python_version'): '%d.%d.%d' % sys.version_info[:3], + Atom('executable'): sys.executable, + Atom('platform'): sys.platform, + Atom('hash_seed'): os.environ.get('PYTHONHASHSEED', 'random'), + } + event((Atom('ready'), info)) + + sel = selectors.DefaultSelector() + sel_fd = sel.fileno() if hasattr(sel, 'fileno') else -1 + sel.register(ctrl, selectors.EVENT_READ, 'ctrl') + sel.register(rpipe, selectors.EVENT_READ, 'chld') + buf = bytearray() + while True: + for key, _ in sel.select(): + if key.data == 'chld': + try: + os.read(rpipe, 4096) + except BlockingIOError: + pass + reap() + continue + data = ctrl.recv(1024 * 1024) + if not data: + # Erlang closed the template: sessions keep their own + # sockets and outlive us + os._exit(0) + buf += data + while len(buf) >= header.size: + frame_id, length = header.unpack_from(buf) + if len(buf) < header.size + length: + break + body = bytes(buf[header.size:header.size + length]) + del buf[:header.size + length] + handle(frame_id, body[0], _etf.decode(body[1:]) if len(body) > 1 else None) + + +if __name__ == '__main__': + main(sys.argv) diff --git a/src/erlang_python_sup.erl b/src/erlang_python_sup.erl index 1a5c459..dc8f192 100644 --- a/src/erlang_python_sup.erl +++ b/src/erlang_python_sup.erl @@ -128,6 +128,16 @@ init([]) -> modules => [py_context_sup] }, + %% Session templates (py_session) + SessionSupSpec = #{ + id => py_session_sup, + start => {py_session_sup, start_link, []}, + restart => permanent, + shutdown => infinity, + type => supervisor, + modules => [py_session_sup] + }, + %% Context router initialization (starts contexts under py_context_sup) ContextRouterInitSpec = #{ id => py_context_init, @@ -179,7 +189,7 @@ init([]) -> }, Children = [CallbackSpec, ShmSpec, ThreadHandlerSpec, LoggerSpec, TracerSpec, - ContextSupSpec, ContextRouterInitSpec, + ContextSupSpec, SessionSupSpec, ContextRouterInitSpec, WorkerRegistrySpec, WorkerSupSpec, EventLoopSpec, EventLoopPoolSpec], diff --git a/src/py_child.erl b/src/py_child.erl index db22e81..62ca7d5 100644 --- a/src/py_child.erl +++ b/src/py_child.erl @@ -87,14 +87,17 @@ resolve_exe(Exe) -> end end. +%% @doc Absolute: a session child starts in its own directory, so a path +%% relative to the VM's cwd (code added with -pa _build/...) would not resolve. -spec priv_dir() -> file:filename(). priv_dir() -> - case code:priv_dir(erlang_python) of + Dir = case code:priv_dir(erlang_python) of {error, bad_name} -> filename:join(filename:dirname(filename:dirname(code:which(?MODULE))), "priv"); - Dir -> - Dir - end. + D -> + D + end, + filename:absname(Dir). %% @doc Private directory for the sockets of this node (mode 0700). Kept %% under `$TMPDIR': a Unix socket path is limited to 104 bytes. diff --git a/src/py_isolated.erl b/src/py_isolated.erl index 70ca250..493bcab 100644 --- a/src/py_isolated.erl +++ b/src/py_isolated.erl @@ -347,10 +347,14 @@ handle_event(info, {kill, From, MRef}, State, Data) -> end; handle_event(info, {stop, From, MRef}, _State, #data{opts = Opts} = Data) -> How = case maps:get(session, Opts, false) of - true -> kill; + %% The zygote (or the VM, for a spawned one) reaps the child: + %% nothing to wait for once it has SIGKILL + true -> kill_nowait; false -> graceful end, Data1 = stop_child(Data, How), + %% Gone before the caller hears back: a closed session leaves nothing + remove_scratch(Data1), From ! {MRef, ok}, {stop, normal, Data1}; @@ -1065,8 +1069,12 @@ child_exited(Reason, State, #data{child = Child, opts = Opts} = Data0) -> [Data0#data.id, FailReason]) end, case maps:get(session, Opts, false) of - true -> {next_state, {exited, FailReason}, Data1}; - false -> restart_or_stop(Reason, Data0, Data1) + true -> + %% No new child will come: kill/1 callers are answered now + [W ! {M, ok} || {W, M} <- Data1#data.kill_waiters], + {next_state, {exited, FailReason}, Data1#data{kill_waiters = []}}; + false -> + restart_or_stop(Reason, Data0, Data1) end. restart_or_stop(Reason, Data0, #data{opts = Opts} = Data1) -> @@ -1123,6 +1131,8 @@ stop_child(#data{child = #child{port = Port, os_pid = OsPid} = Child} = Data, Ho ok -> ok; timeout -> kill_port(Port, OsPid), wait_exit(Child, 2000) end; + kill_nowait -> + kill_port(Port, OsPid); _ -> kill_port(Port, OsPid), wait_exit(Child, 2000) diff --git a/src/py_session.erl b/src/py_session.erl new file mode 100644 index 0000000..2578ef7 --- /dev/null +++ b/src/py_session.erl @@ -0,0 +1,144 @@ +%% Copyright 2026 Benoit Chesneau +%% +%% Licensed under the Apache License, Version 2.0 (the "License"); +%% you may not use this file except in compliance with the License. +%% You may obtain a copy of the License at +%% +%% http://www.apache.org/licenses/LICENSE-2.0 +%% +%% Unless required by applicable law or agreed to in writing, software +%% distributed under the License is distributed on an "AS IS" BASIS, +%% WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +%% See the License for the specific language governing permissions and +%% limitations under the License. + +%%% @doc Isolated sessions: a fresh Python process per session, started from +%%% a prepared template. +%%% +%%% A template holds what every session starts with: interpreter, `sys.path', +%%% imports, preload code, environment, hash seed and limits. Each call to +%%% new/1 gives a new child process in which none of the state of another +%%% session exists; close/1 kills it. +%%% +%%% ``` +%%% {ok, T} = py_session:template(#{paths => [Dir], imports => [orders], +%%% hash_seed => 0}), +%%% {ok, S} = py_session:new(T), +%%% {ok, R} = py_context:call(S, orders, handle, [Event, State]), +%%% ok = py_session:close(S). +%%% ''' +%%% +%%% A session is an isolated context: every py_context function works on it. +%%% It is linked to the process that created it and is never restarted; if +%%% its child dies, requests answer `{error, Reason}' until it is closed. +%%% +%%% Template options: +%%%
      +%%%
    • `start' - `fork' (default): sessions are forked from a zygote that +%%% already ran the imports and preload. `spawn': each session starts +%%% a new interpreter; use it when the template cannot be forked +%%% (threads started at import, Objective-C on macOS).
    • +%%%
    • `python', `paths', `imports', `preload' - the prepared state.
    • +%%%
    • `env' - the environment of every session. Nothing is inherited +%%% from the VM unless `clear_env => false'.
    • +%%%
    • `hash_seed' - PYTHONHASHSEED shared by every session (default +%%% `random', chosen once per zygote).
    • +%%%
    • `zygotes' - zygotes forking in turn (default 1).
    • +%%%
    • `warm' - with `start => spawn', sessions kept started ahead.
    • +%%%
    • `rlimits', `cgroup', `start_timeout', `kill_after' - as for +%%% isolated contexts, applied to each session.
    • +%%%
    +-module(py_session). + +-export([template/1, + stop_template/1, + new/1, + new/2, + close/1, + run/4, + run/5, + refresh/1, + info/1]). + +-type template() :: pid(). +-type session() :: pid(). +-export_type([template/0, session/0]). + +%% @doc Build a template. With `start => fork' this starts the zygotes and +%% runs the imports and preload before returning. +-spec template(map()) -> {ok, template()} | {error, term()}. +template(Opts) when is_map(Opts) -> + py_session_sup:start_template(Opts). + +-spec stop_template(template()) -> ok. +stop_template(T) -> + py_session_sup:stop_template(T). + +%% @doc A new session: a fresh isolated process, linked to the caller. +-spec new(template()) -> {ok, session()} | {error, term()}. +new(T) -> + new(T, #{}). + +%% @doc A new session. `Opts' may carry `timeout' (milliseconds to get it). +-spec new(template(), map()) -> {ok, session()} | {error, term()}. +new(T, Opts) -> + Timeout = maps:get(timeout, Opts, 15000), + case py_session_template:checkout(T, Timeout) of + {ok, Ctx} -> + %% Started ahead by the template: take it over + MRef = erlang:monitor(process, Ctx), + Ctx ! {set_parent, self(), MRef, self()}, + receive + {MRef, ok} -> + erlang:demonitor(MRef, [flush]), + {ok, Ctx}; + {'DOWN', MRef, process, Ctx, _} -> + new(T, Opts) + after Timeout -> + erlang:demonitor(MRef, [flush]), + {error, timeout} + end; + {start, CtxOpts} -> + py_context:new(maps:merge(CtxOpts, maps:with([start_timeout], Opts))); + {error, _} = Err -> + Err + end. + +%% @doc Kill the session's process. A session is never reused. +-spec close(session()) -> ok. +close(S) -> + py_context:stop(S). + +%% @doc Run one call in a new session and close it. +-spec run(template(), atom() | binary(), atom() | binary(), list()) -> + {ok, term()} | {error, term()}. +run(T, Module, Func, Args) -> + run(T, Module, Func, Args, #{}). + +%% @doc Run one call in a new session and close it. `Opts': `kwargs', +%% `timeout' (for the call), plus the options of new/2. +-spec run(template(), atom() | binary(), atom() | binary(), list(), map()) -> + {ok, term()} | {error, term()}. +run(T, Module, Func, Args, Opts) -> + case new(T, Opts) of + {ok, S} -> + try + py_context:call(S, Module, Func, Args, maps:get(kwargs, Opts, #{}), + maps:get(timeout, Opts, infinity)) + after + close(S) + end; + {error, _} = Err -> + Err + end. + +%% @doc Rebuild the template: new zygotes (code is imported again) or new +%% warm sessions. Live sessions are not touched. +-spec refresh(template()) -> ok | {error, term()}. +refresh(T) -> + py_session_template:refresh(T). + +%% @doc What the template runs: start mode, zygotes, sessions alive, forks. +-spec info(template()) -> map(). +info(T) -> + py_session_template:info(T). diff --git a/src/py_session_sup.erl b/src/py_session_sup.erl new file mode 100644 index 0000000..5baad8c --- /dev/null +++ b/src/py_session_sup.erl @@ -0,0 +1,51 @@ +%% Copyright 2026 Benoit Chesneau +%% +%% Licensed under the Apache License, Version 2.0 (the "License"); +%% you may not use this file except in compliance with the License. +%% You may obtain a copy of the License at +%% +%% http://www.apache.org/licenses/LICENSE-2.0 +%% +%% Unless required by applicable law or agreed to in writing, software +%% distributed under the License is distributed on an "AS IS" BASIS, +%% WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +%% See the License for the specific language governing permissions and +%% limitations under the License. + +%%% @doc Supervisor of session templates. +%%% +%%% @private +-module(py_session_sup). + +-behaviour(supervisor). + +-export([start_link/0, start_template/1, stop_template/1]). +-export([init/1]). + +-spec start_link() -> {ok, pid()} | {error, term()}. +start_link() -> + supervisor:start_link({local, ?MODULE}, ?MODULE, []). + +-spec start_template(map()) -> {ok, pid()} | {error, term()}. +start_template(Opts) -> + case supervisor:start_child(?MODULE, [Opts]) of + {ok, Pid} -> {ok, Pid}; + {error, _} = Err -> Err + end. + +-spec stop_template(pid()) -> ok. +stop_template(T) when is_pid(T) -> + _ = supervisor:terminate_child(?MODULE, T), + ok. + +init([]) -> + SupFlags = #{strategy => simple_one_for_one, intensity => 5, period => 10}, + ChildSpec = #{ + id => py_session_template, + start => {py_session_template, start_link, []}, + restart => temporary, + shutdown => 5000, + type => worker, + modules => [py_session_template] + }, + {ok, {SupFlags, [ChildSpec]}}. diff --git a/src/py_session_template.erl b/src/py_session_template.erl new file mode 100644 index 0000000..e03b10b --- /dev/null +++ b/src/py_session_template.erl @@ -0,0 +1,504 @@ +%% Copyright 2026 Benoit Chesneau +%% +%% Licensed under the Apache License, Version 2.0 (the "License"); +%% you may not use this file except in compliance with the License. +%% You may obtain a copy of the License at +%% +%% http://www.apache.org/licenses/LICENSE-2.0 +%% +%% Unless required by applicable law or agreed to in writing, software +%% distributed under the License is distributed on an "AS IS" BASIS, +%% WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +%% See the License for the specific language governing permissions and +%% limitations under the License. + +%%% @doc A prepared Python environment that sessions start from. +%%% +%%% `start => fork' keeps one or more zygotes (priv/py_zygote.py): Python +%%% processes that ran the template's imports and preload once and fork a +%%% fresh child per session. The zygote reports each child's exit here and +%%% the template passes it to the session's context as +%%% `{py_session_exited, OsPid, Code}'. +%%% +%%% `start => spawn' starts each session as a new isolated child and can keep +%%% `warm' of them started ahead of time; each is handed out once. +%%% +%%% A crashed zygote is rebuilt; the sessions it forked are separate +%%% processes and keep running. +%%% +%%% @private +%%% +%%% Owns: the zygote ports and control sockets, or the warm contexts. +%%% Talks to: `py_isolated' (fork requests, exit reports), `py_session'. +%%% Never: runs session requests; those go straight to the session context. +-module(py_session_template). + +-behaviour(gen_server). + +-export([start_link/1, + fork/4, + checkout/2, + refresh/1, + info/1]). + +-export([init/1, handle_call/3, handle_cast/2, handle_info/2, terminate/2]). + +-define(STATUS_REQUEST, 0). +-define(STATUS_ERROR, 1). +-define(STATUS_OK, 2). +-define(STATUS_EVENT, 4). +-define(DEFAULT_START_TIMEOUT_MS, 10000). +-define(MAX_REBUILDS, 5). +-define(REBUILD_PERIOD_MS, 10000). + +-record(zygote, { + port :: port(), + os_pid :: non_neg_integer(), + listener :: socket:socket(), + sock :: socket:socket(), + buf = <<>> :: binary(), + info = #{} :: map() +}). + +-record(st, { + opts :: map(), + start :: fork | spawn, + zygotes = [] :: [#zygote{}], + next = 0 :: non_neg_integer(), + next_id = 1 :: pos_integer(), + %% Id => From of fork requests the zygote has + pending = #{} :: #{pos_integer() => {gen_server:from(), pid()}}, + %% OsPid => context pid of live forked sessions + owners = #{} :: #{pos_integer() => pid()}, + %% start => spawn: contexts started ahead of time + warm = [] :: [pid()], + forks = 0 :: non_neg_integer(), + build_ms = 0 :: non_neg_integer(), + rebuilds = [] :: [integer()] +}). + +%% ============================================================================ +%% API +%% ============================================================================ + +-spec start_link(map()) -> {ok, pid()} | {error, term()}. +start_link(Opts) -> + gen_server:start_link(?MODULE, Opts, []). + +%% @doc Fork a session child that connects to SockPath. Called by the +%% session's context process, which receives the child's exit. +-spec fork(pid(), file:filename(), map(), timeout()) -> {ok, pos_integer()} | {error, term()}. +fork(T, SockPath, ForkOpts, Timeout) -> + try gen_server:call(T, {fork, SockPath, ForkOpts}, Timeout) + catch + exit:{timeout, _} -> {error, fork_timeout}; + exit:{noproc, _} -> {error, template_stopped}; + exit:{Reason, _} -> {error, {template_down, Reason}} + end. + +%% @doc A context started ahead of time (`{ok, Ctx}', linked to the +%% template until handed over), or the options to start one. +-spec checkout(pid(), timeout()) -> {ok, pid()} | {start, map()} | {error, term()}. +checkout(T, Timeout) -> + try gen_server:call(T, checkout, Timeout) + catch + exit:{noproc, _} -> {error, template_stopped}; + exit:{Reason, _} -> {error, {template_down, Reason}} + end. + +-spec refresh(pid()) -> ok | {error, term()}. +refresh(T) -> + gen_server:call(T, refresh, infinity). + +-spec info(pid()) -> map(). +info(T) -> + gen_server:call(T, info). + +%% ============================================================================ +%% gen_server callbacks +%% ============================================================================ + +init(Opts) -> + process_flag(trap_exit, true), + case check_opts(Opts) of + ok -> + St = #st{opts = Opts, start = maps:get(start, Opts, fork)}, + case build(St) of + {ok, St1} -> {ok, St1}; + {error, _} = Err -> Err + end; + {error, _} = Err -> + Err + end. + +handle_call({fork, _Path, _ForkOpts}, _From, #st{start = spawn} = St) -> + {reply, {error, not_a_fork_template}, St}; +handle_call({fork, _Path, _ForkOpts}, _From, #st{zygotes = []} = St) -> + {reply, {error, no_zygote}, St}; +handle_call({fork, Path, ForkOpts}, {Pid, _} = From, St) -> + #st{zygotes = Zs, next = N, next_id = Id, pending = Pending} = St, + Z = lists:nth(N rem length(Zs) + 1, Zs), + Term = {fork, py_child:to_bin(Path), Pid, fork_opts(ForkOpts)}, + case py_child:send_frame(Z#zygote.sock, Id, ?STATUS_REQUEST, Term) of + ok -> + {noreply, St#st{next = N + 1, next_id = Id + 1, + pending = Pending#{Id => {From, Pid}}}}; + {error, Reason} -> + {reply, {error, {zygote_unreachable, Reason}}, St} + end; +handle_call(checkout, _From, #st{start = fork} = St) -> + {reply, {start, context_opts(St)}, St}; +handle_call(checkout, _From, #st{warm = [Ctx | Rest]} = St) -> + self() ! fill_warm, + {reply, {ok, Ctx}, St#st{warm = Rest}}; +handle_call(checkout, _From, #st{warm = []} = St) -> + self() ! fill_warm, + {reply, {start, context_opts(St)}, St}; +handle_call(refresh, _From, #st{start = fork, zygotes = Old} = St) -> + case build(St#st{zygotes = []}) of + {ok, St1} -> + %% The old zygotes exit when their control socket closes; the + %% sessions they forked keep running and are watched by pid + [retire(Z) || Z <- Old], + {reply, ok, St1#st{owners = #{}}}; + {error, Reason} -> + {reply, {error, Reason}, St} + end; +handle_call(refresh, _From, #st{start = spawn, warm = Warm} = St) -> + [begin unlink(C), py_context:stop(C) end || C <- Warm], + self() ! fill_warm, + {reply, ok, St#st{warm = []}}; +handle_call(info, _From, St) -> + {reply, info_map(St), St}; +handle_call(_Other, _From, St) -> + {reply, {error, unknown_request}, St}. + +handle_cast(_Msg, St) -> + {noreply, St}. + +handle_info({'$socket', S, select, _}, St) -> + case lists:keyfind(S, #zygote.sock, St#st.zygotes) of + false -> {noreply, St}; + Z -> {noreply, drain(Z, St)} + end; +handle_info({'$socket', _, _, _}, St) -> + %% abort: the port's exit_status follows and drives the rebuild + {noreply, St}; +handle_info({Port, {data, Out}}, St) when is_port(Port) -> + [logger:info("py_session template ~p: ~s", [self(), L]) + || L <- binary:split(Out, <<"\n">>, [global, trim_all])], + {noreply, St}; +handle_info({Port, {exit_status, Status}}, St) when is_port(Port) -> + case lists:keytake(Port, #zygote.port, St#st.zygotes) of + {value, Z, Rest} -> zygote_exited(Z, py_child:exit_reason(Status), St#st{zygotes = Rest}); + false -> {noreply, St} + end; +handle_info(fill_warm, #st{start = spawn, warm = Warm, opts = Opts} = St) -> + case length(Warm) < maps:get(warm, Opts, 0) of + true -> + case py_context:start_link(erlang:unique_integer([positive]), isolated, + context_opts(St)) of + {ok, Ctx} -> + self() ! fill_warm, + {noreply, St#st{warm = Warm ++ [Ctx]}}; + {error, Reason} -> + logger:warning("py_session template ~p: warm session failed: ~p", + [self(), Reason]), + {noreply, St} + end; + false -> + {noreply, St} + end; +handle_info(fill_warm, St) -> + {noreply, St}; +handle_info({'EXIT', Pid, _Reason}, #st{warm = Warm} = St) -> + %% A warm session died before anyone took it + {noreply, St#st{warm = lists:delete(Pid, Warm)}}; +handle_info(_Msg, St) -> + {noreply, St}. + +terminate(_Reason, #st{zygotes = Zs, warm = Warm}) -> + [retire(Z) || Z <- Zs], + [py_context:stop(C) || C <- Warm], + ok. + +%% ============================================================================ +%% Building +%% ============================================================================ + +check_opts(Opts) -> + Checks = [ + fun() -> case maps:get(start, Opts, fork) of + S when S =:= fork; S =:= spawn -> ok; + S -> {error, {badarg, {start, S}}} + end end, + fun() -> case maps:get(zygotes, Opts, 1) of + N when is_integer(N), N >= 1 -> ok; + N -> {error, {badarg, {zygotes, N}}} + end end, + fun() -> case maps:get(warm, Opts, 0) of + N when is_integer(N), N >= 0 -> ok; + N -> {error, {badarg, {warm, N}}} + end end, + fun() -> py_child:check_env_opts(env_opts(Opts)) end, + fun() -> case {maps:get(cgroup, Opts, undefined), os:type()} of + {undefined, _} -> ok; + {_, {unix, linux}} -> ok; + {_, {unix, Os}} -> {error, {cgroup_unsupported, Os}} + end end, + fun() -> case py_child:python_executable(Opts) of + {error, _} = Err -> Err; + _ -> ok + end end + ], + lists:foldl(fun(Check, ok) -> Check(); (_, Err) -> Err end, ok, Checks). + +%% Sessions see only the template's env unless it asks otherwise +env_opts(Opts) -> + maps:merge(#{clear_env => true}, maps:with([env, clear_env, hash_seed], Opts)). + +build(#st{start = spawn} = St) -> + self() ! fill_warm, + {ok, St}; +build(#st{opts = Opts} = St) -> + T0 = erlang:monotonic_time(millisecond), + case build_zygotes(maps:get(zygotes, Opts, 1), St#st.opts, []) of + {ok, Zs} -> + {ok, St#st{zygotes = Zs, build_ms = erlang:monotonic_time(millisecond) - T0}}; + {error, _} = Err -> + Err + end. + +build_zygotes(0, _Opts, Acc) -> + {ok, lists:reverse(Acc)}; +build_zygotes(N, Opts, Acc) -> + case start_zygote(Opts) of + {ok, Z} -> + build_zygotes(N - 1, Opts, [Z | Acc]); + {error, _} = Err -> + [retire(Z) || Z <- Acc], + Err + end. + +start_zygote(Opts) -> + Python = py_child:python_executable(Opts), + Path = py_child:new_sock_path("tpl_"), + Timeout = maps:get(start_timeout, Opts, ?DEFAULT_START_TIMEOUT_MS), + case py_child:listen(Path) of + {ok, L} -> + Script = filename:join(py_child:priv_dir(), "py_zygote.py"), + Port = open_port({spawn_executable, Python}, + [exit_status, stderr_to_stdout, binary, use_stdio, + {args, [Script, Path]}, + {env, py_child:port_env(env_opts(Opts))}]), + OsPid = case erlang:port_info(Port, os_pid) of + {os_pid, P} -> P; + _ -> 0 + end, + Result = case py_child:accept(L, {port, Port}, Timeout) of + {ok, S} -> + py_child:tune_socket(S), + Z = #zygote{port = Port, os_pid = OsPid, listener = L, sock = S}, + prepare(Z, Opts, Timeout); + {error, _} = Err -> + Err + end, + _ = file:delete(Path), + case Result of + {ok, _} = Ok -> + Ok; + {error, Reason} -> + socket:close(L), + py_child:kill_os_pid(OsPid), + close_port(Port), + {error, {template_failed, Reason}} + end; + {error, Reason} -> + {error, {template_failed, Reason}} + end. + +%% Wait for `ready', run the imports and preload, then arm the socket. +prepare(Z, Opts, Timeout) -> + case recv_sync(Z, Timeout) of + {ok, {0, ?STATUS_EVENT, {ready, Info}}, Z1} -> + Paths = [py_child:to_bin(P) || P <- py_import:all_paths()] + ++ [py_child:to_bin(P) || P <- maps:get(paths, Opts, [])], + Imports = lists:usort([py_child:to_bin(M) || {M, _} <- py_import:all_imports()] + ++ [py_child:to_bin(M) || M <- maps:get(imports, Opts, [])]), + Preload = iolist_to_binary([preload_code(), <<"\n">>, + maps:get(preload, Opts, <<>>)]), + ok = py_child:send_frame(Z1#zygote.sock, 1, ?STATUS_REQUEST, + {init, Paths, Imports, Preload}), + case recv_sync(Z1, Timeout) of + {ok, {1, ?STATUS_OK, _}, Z2} -> + {ok, arm(Z2#zygote{info = Info})}; + {ok, {1, ?STATUS_ERROR, Why}, _} -> + {error, {init_failed, Why}}; + {ok, Other, _} -> + {error, {unexpected_frame, Other}}; + {error, _} = Err -> + Err + end; + {ok, Other, _} -> + {error, {unexpected_frame, Other}}; + {error, _} = Err -> + Err + end. + +preload_code() -> + try py_preload:get_code() of + Code when is_binary(Code) -> Code; + _ -> <<>> + catch + _:_ -> <<>> + end. + +recv_sync(#zygote{sock = S, buf = Buf, port = Port} = Z, Timeout) -> + case py_child:parse_frame(Buf) of + {ok, Frame, Rest} -> + {ok, Frame, Z#zygote{buf = Rest}}; + more -> + case socket:recv(S, 0, Timeout) of + {ok, Data} -> + recv_sync(Z#zygote{buf = <>}, Timeout); + {error, Reason} -> + {error, {Reason, drain_output(Port, [])}} + end; + {error, _} = Err -> + Err + end. + +drain_output(Port, Acc) -> + receive + {Port, {data, D}} -> drain_output(Port, [D | Acc]) + after 50 -> + iolist_to_binary(lists:reverse(Acc)) + end. + +%% Arm the select; frames already buffered are processed first +arm(#zygote{} = Z) -> + self() ! {'$socket', Z#zygote.sock, select, arm}, + Z. + +retire(#zygote{sock = S, listener = L, port = Port}) -> + _ = socket:close(S), + _ = socket:close(L), + close_port(Port). + +close_port(Port) -> + try port_close(Port) catch error:badarg -> ok end, + ok. + +%% ============================================================================ +%% Frames from a zygote +%% ============================================================================ + +drain(#zygote{sock = S, buf = Buf} = Z, St) -> + case socket:recv(S, 0, nowait) of + {ok, Data} -> + drain(Z#zygote{buf = <>}, St); + {select, _} -> + frames(Z, St); + {error, _} -> + %% The port's exit_status follows + frames(Z, St) + end. + +frames(#zygote{buf = Buf} = Z, St) -> + case py_child:parse_frame(Buf) of + {ok, Frame, Rest} -> + St1 = frame(Frame, St), + frames(Z#zygote{buf = Rest}, St1); + more -> + store(Z, St); + {error, Reason} -> + logger:error("py_session template ~p: bad frame from zygote ~p: ~p", + [self(), Z#zygote.os_pid, Reason]), + py_child:kill_os_pid(Z#zygote.os_pid), + store(Z#zygote{buf = <<>>}, St) + end. + +store(Z, #st{zygotes = Zs} = St) -> + St#st{zygotes = lists:keyreplace(Z#zygote.sock, #zygote.sock, Zs, Z)}. + +frame({Id, Status, Term}, #st{pending = Pending, owners = Owners} = St) + when Status =:= ?STATUS_OK; Status =:= ?STATUS_ERROR -> + case maps:take(Id, Pending) of + {{From, Pid}, Rest} when Status =:= ?STATUS_OK -> + gen_server:reply(From, {ok, Term}), + St#st{pending = Rest, owners = Owners#{Term => Pid}, forks = St#st.forks + 1}; + {{From, _Pid}, Rest} -> + gen_server:reply(From, {error, {fork_failed, Term}}), + St#st{pending = Rest}; + error -> + St + end; +frame({_, ?STATUS_EVENT, {exited, OsPid, Code}}, #st{owners = Owners} = St) -> + case maps:take(OsPid, Owners) of + {Pid, Rest} -> + Pid ! {py_session_exited, OsPid, Code}, + St#st{owners = Rest}; + error -> + St + end; +frame(_, St) -> + St. + +%% ============================================================================ +%% Zygote failure +%% ============================================================================ + +zygote_exited(Z, Reason, #st{pending = Pending} = St) -> + logger:warning("py_session template ~p: zygote ~p exited: ~p", + [self(), Z#zygote.os_pid, Reason]), + retire(Z), + %% Forks it had not answered fail; the sessions it forked keep running + %% and their contexts watch them by pid + [gen_server:reply(From, {error, {zygote_exited, Reason}}) || {From, _} <- maps:values(Pending)], + St1 = St#st{pending = #{}}, + Now = erlang:monotonic_time(millisecond), + Recent = [T || T <- St1#st.rebuilds, Now - T =< ?REBUILD_PERIOD_MS], + case length(Recent) < ?MAX_REBUILDS of + true -> + case start_zygote(St1#st.opts) of + {ok, NewZ} -> + {noreply, St1#st{zygotes = St1#st.zygotes ++ [NewZ], + rebuilds = [Now | Recent]}}; + {error, Why} -> + {stop, {zygote_rebuild_failed, Why}, St1} + end; + false -> + {stop, {zygote_exited, Reason}, St1} + end. + +%% ============================================================================ +%% Helpers +%% ============================================================================ + +%% Options of the isolated context behind each session +context_opts(#st{start = fork, opts = Opts}) -> + maps:merge(maps:with([rlimits, cgroup, start_timeout, kill_after], Opts), + #{mode => isolated, session => true, restart => false, + origin => {fork, self()}}); +context_opts(#st{start = spawn, opts = Opts}) -> + maps:merge(maps:with([python, paths, preload, rlimits, cgroup, start_timeout, + kill_after], Opts), + (env_opts(Opts))#{mode => isolated, session => true, restart => false, + origin => spawn}). + +fork_opts(ForkOpts) -> + maps:fold(fun(rlimits, V, Acc) -> Acc#{rlimits => V}; + (cgroup, V, Acc) -> Acc#{cgroup => py_child:to_bin(V)}; + (cd, V, Acc) -> Acc#{cd => py_child:to_bin(V)}; + (_, _, Acc) -> Acc + end, #{}, ForkOpts). + +info_map(#st{start = Start, zygotes = Zs, owners = Owners, forks = Forks, + build_ms = BuildMs, warm = Warm, opts = Opts}) -> + #{start => Start, + zygotes => [Info#{os_pid => P} || #zygote{os_pid = P, info = Info} <- Zs], + sessions => map_size(Owners), + forks => Forks, + build_ms => BuildMs, + warm => length(Warm), + hash_seed => maps:get(hash_seed, Opts, random)}. diff --git a/test/py_session_SUITE.erl b/test/py_session_SUITE.erl new file mode 100644 index 0000000..0b96ca0 --- /dev/null +++ b/test/py_session_SUITE.erl @@ -0,0 +1,545 @@ +%%% @doc Isolated sessions (py_session): a fresh process per session, started +%%% from a template, by fork from a zygote or by spawning a child. +%%% +%%% The `fork' and `spawn' groups run the same cases; `fork_only' and +%%% `spawn_only' cover what differs. +-module(py_session_SUITE). + +-include_lib("common_test/include/ct.hrl"). + +-export([all/0, groups/0, init_per_suite/1, end_per_suite/1, + init_per_group/2, end_per_group/2, end_per_testcase/2]). + +%% logger handler used by test_print_is_logged +-export([log/2]). + +-export([ + test_sessions_are_isolated/1, + test_same_start_state/1, + test_env_replaced/1, + test_hash_seed/1, + test_random_differs/1, + test_context_api/1, + test_reentrance_depth/1, + test_concurrent_reentrance/1, + test_cross_session_call/1, + test_interrupt_nested_call/1, + test_session_crash/1, + test_kill_session/1, + test_close_reaps_and_cleans/1, + test_caller_crash_stops_session/1, + test_run_helper/1, + test_print_is_logged/1, + test_rlimits_apply/1, + test_refresh_picks_new_code/1, + test_bad_template_options/1, + test_zygote_crash_rebuilds/1, + test_thread_at_import_refused/1, + test_erlang_call_during_preload/1, + test_info_counts_forks/1, + test_warm_pool/1, + test_thread_at_import_spawned/1 +]). + +-define(MOD, py_test_session). + +all() -> + [{group, fork}, {group, spawn}, {group, fork_only}, {group, spawn_only}]. + +groups() -> + Common = [ + test_sessions_are_isolated, + test_same_start_state, + test_env_replaced, + test_hash_seed, + test_random_differs, + test_context_api, + test_reentrance_depth, + test_concurrent_reentrance, + test_cross_session_call, + test_interrupt_nested_call, + test_session_crash, + test_kill_session, + test_close_reaps_and_cleans, + test_caller_crash_stops_session, + test_run_helper, + test_print_is_logged, + test_rlimits_apply, + test_refresh_picks_new_code, + test_bad_template_options + ], + [{fork, [], Common}, + {spawn, [], Common}, + {fork_only, [], [test_zygote_crash_rebuilds, + test_thread_at_import_refused, + test_erlang_call_during_preload, + test_info_counts_forks]}, + {spawn_only, [], [test_warm_pool, + test_thread_at_import_spawned]}]. + +init_per_suite(Config) -> + {ok, _} = application:ensure_all_started(erlang_python), + TestDir = filename:join(code:lib_dir(erlang_python), "test"), + py:register_function(sess_reenter, fun([Ctx, N]) -> + {ok, R} = py_context:call(Ctx, ?MOD, reenter, [N - 1]), + R + end), + py:register_function(sess_nested_sleep, fun([Ctx, Seconds]) -> + py_context:call(Ctx, ?MOD, sleep, [Seconds], #{}, 200) + end), + [{test_dir, TestDir} | Config]. + +end_per_suite(_Config) -> + py:unregister_function(sess_reenter), + py:unregister_function(sess_nested_sleep), + ok. + +init_per_group(G, Config) when G =:= fork; G =:= fork_only -> + [{start, fork} | Config]; +init_per_group(_G, Config) -> + [{start, spawn} | Config]. + +end_per_group(_G, _Config) -> + ok. + +end_per_testcase(_Case, _Config) -> + [py_session:stop_template(T) || {_, T, _, _} <- supervisor:which_children(py_session_sup)], + ok. + +%%% ============================================================================ +%%% Common cases +%%% ============================================================================ + +%% Nothing a session changes is seen by the next one: module globals, +%% sys.modules, os.environ, threads, files in its directory. +test_sessions_are_isolated(Config) -> + T = template(Config), + Prev = lists:foldl(fun(I, Prev) -> + {ok, S} = py_session:new(T), + Mark = integer_to_binary(I), + {ok, Prev} = py_context:call(S, ?MOD, peek, []), + {ok, Mark} = py_context:call(S, ?MOD, mark, [Mark]), + py_session:close(S), + none + end, none, lists:seq(1, 100)), + none = Prev, + {ok, A} = py_session:new(T), + {ok, true} = py_context:call(A, ?MOD, put_module, [<<"sess_marker">>]), + {ok, <<"1">>} = py_context:call(A, ?MOD, env_set, [<<"SESS_WRITE">>, <<"1">>]), + {ok, N0} = py_context:call(A, ?MOD, thread_count, []), + {ok, _} = py_context:call(A, ?MOD, start_thread, []), + {ok, [<<"f">>]} = py_context:call(A, ?MOD, write_file, [<<"f">>, <<"x">>]), + {ok, B} = py_session:new(T), + {ok, false} = py_context:call(B, ?MOD, has_module, [<<"sess_marker">>]), + {ok, none} = py_context:call(B, ?MOD, env_get, [<<"SESS_WRITE">>]), + {ok, N0} = py_context:call(B, ?MOD, thread_count, []), + {ok, []} = py_context:call(B, ?MOD, list_cwd, []), + {ok, PA} = py_context:call(A, ?MOD, pid, []), + {ok, PB} = py_context:call(B, ?MOD, pid, []), + true = PA =/= PB, + py_session:close(A), + py_session:close(B). + +%% Every session starts with the template's imports and preload; with fork +%% they ran once, in the zygote. +test_same_start_state(Config) -> + T = template(Config, #{preload => <<"PRELOADED = 'yes'">>}), + {ok, A} = py_session:new(T), + {ok, B} = py_session:new(T), + {ok, <<"yes">>} = py_context:eval(A, <<"PRELOADED">>), + {ok, <<"yes">>} = py_context:eval(B, <<"PRELOADED">>), + {ok, ImportedA} = py_context:call(A, ?MOD, imported_in, []), + {ok, ImportedB} = py_context:call(B, ?MOD, imported_in, []), + {ok, PidA} = py_context:call(A, ?MOD, pid, []), + case ?config(start, Config) of + fork -> + %% imported once, by the zygote, before either session existed + ImportedA = ImportedB, + true = ImportedA =/= PidA; + spawn -> + ImportedA = PidA + end, + py_session:close(A), + py_session:close(B). + +%% The VM's environment does not reach a session: it sees the template's env. +test_env_replaced(Config) -> + true = os:putenv("SESS_VM_ONLY", "leak"), + try + T = template(Config, #{env => #{"SESS_DECLARED" => "1"}}), + {ok, S} = py_session:new(T), + {ok, none} = py_context:call(S, ?MOD, env_get, [<<"SESS_VM_ONLY">>]), + {ok, <<"1">>} = py_context:call(S, ?MOD, env_get, [<<"SESS_DECLARED">>]), + {ok, Names} = py_context:call(S, ?MOD, env_names, []), + false = lists:member(<<"HOME">>, Names), + py_session:close(S) + after + os:unsetenv("SESS_VM_ONLY") + end. + +%% One hash seed per template: every session hashes strings and orders sets +%% the same way; another seed changes it. +test_hash_seed(Config) -> + T1 = template(Config, #{hash_seed => 11}), + T2 = template(Config, #{hash_seed => 12}), + Probe = fun(T) -> + {ok, S} = py_session:new(T), + {ok, R} = py_context:call(S, ?MOD, hash_probe, []), + py_session:close(S), + R + end, + P = Probe(T1), + P = Probe(T1), + true = P =/= Probe(T2), + ok. + +%% A forked session does not inherit the zygote's random state as is. +test_random_differs(Config) -> + T = template(Config), + Values = [begin + {ok, S} = py_session:new(T), + {ok, V} = py_context:call(S, ?MOD, random_value, []), + py_session:close(S), + V + end || _ <- lists:seq(1, 5)], + 5 = length(lists:usort(Values)), + ok. + +%% A session is an isolated context: the whole context API works on it. +test_context_api(Config) -> + T = template(Config), + {ok, S} = py_session:new(T), + {ok, 4} = py_context:eval(S, <<"2 + 2">>), + ok = py_context:exec(S, <<"x = 21">>), + {ok, 42} = py_context:eval(S, <<"x * 2">>), + {ok, 3} = py_context:call(S, ?MOD, async_add, [1, 2]), + %% timeout interrupts the call; the session keeps working + {error, timeout} = py_context:call(S, ?MOD, sleep, [30], #{}, 300), + {ok, 1} = py_context:eval(S, <<"1">>), + {ok, LSock} = gen_tcp:listen(0, [{ip, {127, 0, 0, 1}}]), + {ok, Fd} = inet:getfd(LSock), + {ok, ChildFd} = py_context:pass_fd(S, Fd), + {ok, true} = py_context:call(S, ?MOD, fd_is_open, [ChildFd]), + gen_tcp:close(LSock), + ok = py_context:start_loop(S), + {ok, 7} = py_context:submit_await(S, ?MOD, async_add, [3, 4]), + ok = py_context:stop_loop(S), + {ok, Info} = py_context:child_info(S), + true = is_integer(maps:get(os_pid, Info)), + py_session:close(S). + +%% Python -> Erlang -> the same session -> Erlang -> ... ten levels deep. +test_reentrance_depth(Config) -> + T = template(Config), + {ok, S} = py_session:new(T), + {ok, 10} = py_context:call(S, ?MOD, reenter, [10]), + {ok, 1} = py_context:eval(S, <<"1">>), + py_session:close(S). + +test_concurrent_reentrance(Config) -> + T = template(Config), + Self = self(), + Pids = [spawn_link(fun() -> + {ok, S} = py_session:new(T), + R = [py_context:call(S, ?MOD, reenter, [5]) || _ <- lists:seq(1, 10)], + py_session:close(S), + Self ! {self(), R} + end) || _ <- lists:seq(1, 16)], + [receive {P, R} -> R = lists:duplicate(10, {ok, 5}) after 30000 -> ct:fail(timeout) end + || P <- Pids], + ok. + +%% A callback of session A calls into session B. +test_cross_session_call(Config) -> + T = template(Config), + {ok, A} = py_session:new(T), + {ok, B} = py_session:new(T), + {ok, <<"b">>} = py_context:call(B, ?MOD, mark, [<<"b">>]), + py:register_function(sess_cross, fun([_N]) -> + {ok, V} = py_context:call(B, ?MOD, peek, []), + V + end), + try + {ok, <<"b">>} = py_context:call(A, ?MOD, cross, [1]), + {ok, none} = py_context:call(A, ?MOD, peek, []) + after + py:unregister_function(sess_cross) + end, + py_session:close(A), + py_session:close(B). + +%% A nested call that times out is interrupted alone; the outer call it +%% runs inside of completes. +test_interrupt_nested_call(Config) -> + T = template(Config), + {ok, S} = py_session:new(T), + %% {error, timeout} crossed into Python and back: atoms are strings there + {ok, {<<"outer-done">>, {<<"error">>, <<"timeout">>}}} = + py_context:call(S, ?MOD, nested_sleep_then, [30]), + {ok, 1} = py_context:eval(S, <<"1">>), + py_session:close(S). + +%% A session whose child dies answers with the reason until closed; the +%% template and other sessions are untouched. +test_session_crash(Config) -> + T = template(Config), + {ok, A} = py_session:new(T), + {ok, B} = py_session:new(T), + {error, {child_exited, {signal, 6}}} = py_context:call(A, ?MOD, abort, []), + {error, {child_exited, {signal, 6}}} = py_context:eval(A, <<"1">>), + true = is_process_alive(A), + {ok, 1} = py_context:eval(B, <<"1">>), + {ok, C} = py_session:new(T), + {ok, 1} = py_context:eval(C, <<"1">>), + [py_session:close(X) || X <- [A, B, C]], + false = is_process_alive(A), + ok. + +test_kill_session(Config) -> + T = template(Config), + {ok, S} = py_session:new(T), + {ok, Pid} = py_context:call(S, ?MOD, pid, []), + ok = py_context:kill(S), + {error, killed} = py_context:eval(S, <<"1">>), + ok = wait_gone(Pid), + py_session:close(S). + +%% close/1 kills the child (reaped, no zombie) and removes its directory. +test_close_reaps_and_cleans(Config) -> + T = template(Config), + {ok, S} = py_session:new(T), + {ok, Pid} = py_context:call(S, ?MOD, pid, []), + {ok, Cwd} = py_context:call(S, ?MOD, cwd, []), + true = filelib:is_dir(Cwd), + ok = py_session:close(S), + ok = wait_gone(Pid), + false = filelib:is_dir(Cwd), + ok. + +%% A session is linked to the process that created it. +test_caller_crash_stops_session(Config) -> + T = template(Config), + Self = self(), + {Owner, Mon} = spawn_monitor(fun() -> + {ok, S} = py_session:new(T), + {ok, Pid} = py_context:call(S, ?MOD, pid, []), + Self ! {session, S, Pid}, + receive never -> ok end + end), + {S, Pid} = receive {session, S0, P0} -> {S0, P0} after 10000 -> ct:fail(no_session) end, + SMon = erlang:monitor(process, S), + exit(Owner, crash), + receive {'DOWN', Mon, process, Owner, crash} -> ok end, + receive {'DOWN', SMon, process, S, _} -> ok after 5000 -> ct:fail(session_survived) end, + ok = wait_gone(Pid). + +test_run_helper(Config) -> + T = template(Config), + {ok, 42} = py_session:run(T, builtins, int, [<<"42">>]), + {ok, none} = py_session:run(T, ?MOD, peek, []), + {error, {'ValueError', _}} = py_session:run(T, builtins, int, [<<"x">>]), + %% nothing left behind + #{sessions := 0} = wait_sessions(T, 0), + ok. + +%% print() in a session ends up in the Erlang logger: through the port for a +%% spawned child, as log events for a forked one (whose stdio is detached). +test_print_is_logged(Config) -> + #{level := Level} = logger:get_primary_config(), + ok = logger:set_primary_config(level, info), + ok = logger:add_handler(sess_capture, ?MODULE, #{config => #{pid => self()}}), + try + T = template(Config), + {ok, S} = py_session:new(T), + {ok, none} = py_context:eval(S, <<"print('hello-from-session', flush=True)">>), + ok = wait_logged(<<"hello-from-session">>, 50), + py_session:close(S) + after + logger:remove_handler(sess_capture), + logger:set_primary_config(level, Level) + end. + +log(#{msg := Msg}, #{config := #{pid := Pid}}) -> + Text = case Msg of + {string, S} -> S; + {report, R} -> io_lib:format("~p", [R]); + {Format, Args} -> io_lib:format(Format, Args) + end, + Pid ! {logged, unicode:characters_to_binary(Text)}, + ok. + +wait_logged(_Needle, 0) -> + ct:fail(not_logged); +wait_logged(Needle, N) -> + receive + {logged, Text} -> + case binary:match(Text, Needle) of + nomatch -> wait_logged(Needle, N); + _ -> ok + end + after 100 -> + wait_logged(Needle, N - 1) + end. + +test_rlimits_apply(Config) -> + T = template(Config, #{rlimits => #{nofile => 64}}), + {ok, S} = py_session:new(T), + {ok, 64} = py_context:eval(S, <<"__import__('resource').getrlimit(__import__('resource').RLIMIT_NOFILE)[0]">>), + py_session:close(S). + +%% refresh/1 prepares the template again: new sessions see the new code, +%% live ones keep what they started with. +test_refresh_picks_new_code(Config) -> + Dir = filename:join(?config(priv_dir, Config), atom_to_list(?config(start, Config))), + ok = filelib:ensure_dir(filename:join(Dir, "x")), + File = filename:join(Dir, "sess_versioned.py"), + ok = file:write_file(File, <<"VERSION = 1\n">>), + T = template(Config, #{paths => [Dir], imports => [sess_versioned]}), + {ok, Old} = py_session:new(T), + {ok, 1} = py_context:eval(Old, <<"__import__('sess_versioned').VERSION">>), + ok = file:write_file(File, <<"VERSION = 2\n">>), + _ = file:del_dir_r(filename:join(Dir, "__pycache__")), + ok = py_session:refresh(T), + {ok, New} = py_session:new(T), + {ok, 2} = py_context:eval(New, <<"__import__('sess_versioned').VERSION">>), + {ok, 1} = py_context:eval(Old, <<"__import__('sess_versioned').VERSION">>), + py_session:close(Old), + py_session:close(New). + +test_bad_template_options(Config) -> + Start = ?config(start, Config), + {error, {badarg, {start, nope}}} = py_session:template(#{start => nope}), + {error, {badarg, {hash_seed, -1}}} = py_session:template(#{start => Start, hash_seed => -1}), + {error, {python_not_found, _}} = py_session:template(#{start => Start, python => "/no/such/python"}), + case Start of + fork -> + {error, {template_failed, {init_failed, {'ModuleNotFoundError', _}}}} = + py_session:template(#{start => fork, imports => [no_such_module_xyz]}); + spawn -> + ok + end, + ok. + +%%% ============================================================================ +%%% fork only +%%% ============================================================================ + +%% A zygote that dies is rebuilt; the sessions it forked keep working. +test_zygote_crash_rebuilds(Config) -> + T = template(Config), + {ok, S} = py_session:new(T), + #{zygotes := [#{os_pid := Z1}]} = py_session:info(T), + _ = py_nif:os_kill(Z1, 9), + Z2 = wait_new_zygote(T, Z1, 50), + true = Z2 =/= Z1, + {ok, 1} = py_context:eval(S, <<"1">>), + {ok, Pid} = py_context:call(S, ?MOD, pid, []), + {ok, S2} = py_session:new(T), + {ok, 2} = py_context:eval(S2, <<"2">>), + %% the orphaned session is still reaped when closed + py_session:close(S), + ok = wait_gone(Pid), + py_session:close(S2). + +%% A thread started at import would be lost in every fork: refused. +test_thread_at_import_refused(Config) -> + {error, {template_failed, {init_failed, {threads, [<<"import-time-thread">>]}}}} = + py_session:template(#{start => fork, paths => [?config(test_dir, Config)], + imports => [py_test_session_thread]}), + ok. + +%% `erlang' can be imported by template code, but not called before a +%% session exists. +test_erlang_call_during_preload(Config) -> + T = template(Config, #{preload => + <<"import py_test_session\nPRELOAD_CALL = py_test_session.call_during_preload()">>}), + {ok, S} = py_session:new(T), + {ok, Msg} = py_context:eval(S, <<"PRELOAD_CALL">>), + {match, _} = re:run(Msg, <<"not connected">>), + py_session:close(S). + +test_info_counts_forks(Config) -> + T = template(Config, #{zygotes => 2}), + #{start := fork, zygotes := [_, _], forks := 0} = py_session:info(T), + Ss = [begin {ok, S} = py_session:new(T), S end || _ <- lists:seq(1, 4)], + #{forks := 4, sessions := 4} = py_session:info(T), + [py_session:close(S) || S <- Ss], + #{sessions := 0} = wait_sessions(T, 0), + ok. + +%%% ============================================================================ +%%% spawn only +%%% ============================================================================ + +test_warm_pool(Config) -> + T = template(Config, #{warm => 2}), + #{warm := 2} = wait_warm(T, 2, 100), + {Us, {ok, S}} = timer:tc(fun() -> py_session:new(T) end), + ct:pal("session from the warm pool in ~p us", [Us]), + {links, Links} = process_info(S, links), + true = lists:member(self(), Links), + false = lists:member(T, Links), + {ok, 1} = py_context:eval(S, <<"1">>), + #{warm := 2} = wait_warm(T, 2, 100), + py_session:close(S). + +%% spawn starts each session from scratch, so import-time threads are fine. +test_thread_at_import_spawned(Config) -> + T = template(Config, #{imports => [py_test_session_thread]}), + {ok, S} = py_session:new(T), + {ok, 2} = py_context:call(S, ?MOD, thread_count, []), + py_session:close(S). + +%%% ============================================================================ +%%% Helpers +%%% ============================================================================ + +template(Config) -> + template(Config, #{}). + +template(Config, Extra) -> + TestDir = ?config(test_dir, Config), + Paths = [TestDir | maps:get(paths, Extra, [])], + Imports = [?MOD | maps:get(imports, Extra, [])], + Opts = maps:merge(#{start => ?config(start, Config)}, + Extra#{paths => Paths, imports => Imports}), + {ok, T} = py_session:template(Opts), + T. + +wait_gone(Pid) -> + wait_gone(Pid, 100). + +wait_gone(Pid, 0) -> + ct:fail({still_alive, Pid}); +wait_gone(Pid, N) -> + case py_child:os_pid_alive(Pid) of + false -> ok; + true -> timer:sleep(20), wait_gone(Pid, N - 1) + end. + +wait_new_zygote(T, Old, 0) -> + ct:fail({zygote_not_rebuilt, Old, py_session:info(T)}); +wait_new_zygote(T, Old, N) -> + case py_session:info(T) of + #{zygotes := [#{os_pid := P}]} when P =/= Old -> P; + _ -> timer:sleep(50), wait_new_zygote(T, Old, N - 1) + end. + +wait_sessions(T, Want) -> + wait_sessions(T, Want, 100). + +wait_sessions(T, _Want, 0) -> + py_session:info(T); +wait_sessions(T, Want, N) -> + case py_session:info(T) of + #{sessions := Want} = I -> I; + _ -> timer:sleep(20), wait_sessions(T, Want, N - 1) + end. + +wait_warm(T, _Want, 0) -> + py_session:info(T); +wait_warm(T, Want, N) -> + case py_session:info(T) of + #{warm := Want} = I -> I; + _ -> timer:sleep(50), wait_warm(T, Want, N - 1) + end. diff --git a/test/py_test_session.py b/test/py_test_session.py new file mode 100644 index 0000000..f031e54 --- /dev/null +++ b/test/py_test_session.py @@ -0,0 +1,133 @@ +"""Helpers for py_session_SUITE. Imported by the template, so everything at +module level here runs once, in the zygote (start => fork) or in each child +(start => spawn).""" + +import os +import random +import sys +import threading +import time + +import erlang +from erlang import call # resolved before any session exists + +STATE = {} +IMPORTED_IN = os.getpid() + + +def mark(value): + STATE['mark'] = value + return value + + +def peek(): + return STATE.get('mark') + + +def pid(): + return os.getpid() + + +def cwd(): + return os.getcwd() + + +def env_names(): + return sorted(os.environ) + + +def env_get(name): + return os.environ.get(name) + + +def env_set(name, value): + os.environ[name] = value + return value + + +def put_module(name): + import types + sys.modules[name] = types.ModuleType(name) + return name in sys.modules + + +def has_module(name): + return name in sys.modules + + +def start_thread(): + threading.Thread(target=time.sleep, args=(60,), daemon=True).start() + return threading.active_count() + + +def thread_count(): + return threading.active_count() + + +def write_file(name, data): + with open(name, 'w') as f: + f.write(data) + return sorted(os.listdir('.')) + + +def list_cwd(): + return sorted(os.listdir('.')) + + +def random_value(): + return random.random() + + +def hash_probe(): + return (hash('erlang-python'), repr(list({'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h'}))) + + +def reenter(n): + """Python -> Erlang -> this same session -> ... n levels deep.""" + return 0 if n == 0 else call('sess_reenter', erlang.self(), n) + 1 + + +def cross(n): + """Calls into another session through Erlang.""" + return call('sess_cross', n) + + +def nested_sleep_then(seconds): + """Erlang runs sleep() in this session with a short timeout: the nested + request is interrupted, this outer one carries on.""" + inner = call('sess_nested_sleep', erlang.self(), seconds) + return ('outer-done', inner) + + +def sleep(seconds): + time.sleep(seconds) + return 'slept' + + +async def async_add(a, b): + return a + b + + +def fd_is_open(fd): + os.fstat(fd) + return True + + +def imported_in(): + return IMPORTED_IN + + +def call_during_preload(): + try: + call('anything') + except RuntimeError as exc: + return str(exc) + return 'no error' + + +def abort(): + os.abort() + + +def open_files(): + return len(os.listdir('/dev/fd')) diff --git a/test/py_test_session_thread.py b/test/py_test_session_thread.py new file mode 100644 index 0000000..45ca81c --- /dev/null +++ b/test/py_test_session_thread.py @@ -0,0 +1,6 @@ +"""Starts a thread at import time: a template importing it cannot be forked.""" + +import threading +import time + +threading.Thread(target=time.sleep, args=(60,), name='import-time-thread', daemon=True).start() From 52155e3d33ab9008f610cc93507d99b3e6294d9e Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Sat, 26 Sep 2026 10:59:32 +0200 Subject: [PATCH 05/18] Measure sessions, and soak them for leaks The stress suite logs what a fresh session costs next to a plain isolated context, where a fork's time goes, sessions per second and call overhead. The soak suite churns sessions from fork and spawn templates with crashes, kills and timeouts, then checks that processes, ports, fds, children, zombies and scratch directories are back to baseline. The examples compare the isolation step with Temporal's and Restate's Python SDKs on the same workflow. --- examples/README.md | 10 ++ examples/bench_sessions.erl | 114 +++++++++++++++++ examples/bench_sessions_sdks.py | 98 +++++++++++++++ test/py_session_soak_SUITE.erl | 144 ++++++++++++++++++++++ test/py_session_stress_SUITE.erl | 202 +++++++++++++++++++++++++++++++ 5 files changed, 568 insertions(+) create mode 100644 examples/bench_sessions.erl create mode 100644 examples/bench_sessions_sdks.py create mode 100644 test/py_session_soak_SUITE.erl create mode 100644 test/py_session_stress_SUITE.erl diff --git a/examples/README.md b/examples/README.md index 977ac38..e040ca2 100644 --- a/examples/README.md +++ b/examples/README.md @@ -127,6 +127,16 @@ Reactor buffer performance. escript examples/bench_reactor_buffer.erl ``` +### bench_sessions.erl +Isolated sessions: a fresh session (fork, spawn, warm pool) against a plain +isolated context, sessions per second, calls inside a session. +`bench_sessions_sdks.py` measures the isolation step of Temporal's and +Restate's Python SDKs on the same workflow module. +```bash +escript examples/bench_sessions.erl +python3 examples/bench_sessions_sdks.py # needs temporalio, restate-sdk, pydantic +``` + ### bench_resource_pool.erl Resource pool benchmark. ```bash diff --git a/examples/bench_sessions.erl b/examples/bench_sessions.erl new file mode 100644 index 0000000..f10219a --- /dev/null +++ b/examples/bench_sessions.erl @@ -0,0 +1,114 @@ +#!/usr/bin/env escript +%% -*- erlang -*- +%%! -pa _build/default/lib/erlang_python/ebin + +%%% @doc Benchmark for isolated sessions (py_session). +%%% +%%% Measures, with the same workflow module for every row: +%%% 1. a fresh session (new + first call + close): fork template, spawn +%%% template, spawn template with a warm pool, plain isolated context +%%% 2. sessions per second with 1, 8 and 64 callers +%%% 3. a call and a re-entrant call (Python -> Erlang -> same session) +%%% +%%% examples/bench_sessions_sdks.py measures what Temporal's workflow +%%% sandbox and Restate's per-invocation state cost on the same module, to +%%% compare the isolation step of each. +%%% +%%% Run with: +%%% rebar3 compile && escript examples/bench_sessions.erl + +-mode(compile). + +-define(PY, <<" +import sys, types, erlang +m = types.ModuleType('bench_sessions_wf'); sys.modules['bench_sessions_wf'] = m +exec(''' +from dataclasses import dataclass +import json, decimal, datetime + +@dataclass +class Step: + n: int + +STATE = {} + +def handle(event, state): + n = state.get('n', 0) + 1 + return {'commands': [['activity', 'charge', {'n': n}]], 'state': {'n': Step(n).n}} + +def reenter(n): + import erlang + return 0 if n == 0 else erlang.call('bench_reenter', erlang.self(), n) + 1 +''', m.__dict__) +">>). + +main(_) -> + {ok, _} = application:ensure_all_started(erlang_python), + py:register_function(bench_reenter, fun([Ctx, N]) -> + {ok, R} = py_context:call(Ctx, bench_sessions_wf, reenter, [N - 1]), R + end), + Fork = template(#{start => fork}), + Spawn = template(#{start => spawn}), + Warm = template(#{start => spawn, warm => 4}), + io:format("~n== fresh session: new + first call + close ==~n"), + row("fork", session_loop(Fork, 200)), + row("spawn", session_loop(Spawn, 20)), + row("spawn, warm pool", [begin timer:sleep(80), hd(session_loop(Warm, 1)) end + || _ <- lists:seq(1, 20)]), + row("plain isolated", [plain() || _ <- lists:seq(1, 20)]), + io:format("~n== sessions per second, fork template ==~n"), + [begin + L = throughput(Fork, C, 3), + io:format(" ~2w callers ~7.1f sessions/s p50 ~6.2f ms~n", [C, length(L) / 3, p50(L)]) + end || C <- [1, 8, 64]], + io:format("~n== calls inside a session ==~n"), + {ok, S} = py_session:new(Fork), + row("call", [us(fun() -> {ok, _} = py_context:call(S, bench_sessions_wf, handle, [[], #{}]) end) + || _ <- lists:seq(1, 2000)]), + row("re-entrant call", [us(fun() -> {ok, 1} = py_context:call(S, bench_sessions_wf, reenter, [1]) end) + || _ <- lists:seq(1, 1000)]), + py_session:close(S), + ok. + +template(Opts) -> + {ok, T} = py_session:template(Opts#{preload => ?PY}), + T. + +session_loop(T, N) -> + [us(fun() -> + {ok, S} = py_session:new(T), + {ok, _} = py_context:call(S, bench_sessions_wf, handle, [[], #{}]), + py_session:close(S) + end) || _ <- lists:seq(1, N)]. + +plain() -> + us(fun() -> + {ok, C} = py_context:new(#{mode => isolated, preload => ?PY}), + {ok, _} = py_context:call(C, bench_sessions_wf, handle, [[], #{}]), + py_context:stop(C) + end). + +throughput(T, Callers, Secs) -> + Self = self(), + Deadline = erlang:monotonic_time(millisecond) + Secs * 1000, + Loop = fun L(Acc) -> + case erlang:monotonic_time(millisecond) < Deadline of + true -> L([hd(session_loop(T, 1)) | Acc]); + false -> Acc + end + end, + Pids = [spawn_link(fun() -> Self ! {done, self(), Loop([])} end) || _ <- lists:seq(1, Callers)], + lists:append([receive {done, P, L} -> L end || P <- Pids]). + +us(F) -> + T0 = erlang:monotonic_time(microsecond), + _ = F(), + erlang:monotonic_time(microsecond) - T0. + +row(Name, L) -> + S = lists:sort(L), + P = fun(Q) -> lists:nth(max(1, round(Q * length(S))), S) / 1000 end, + io:format(" ~-18s p50 ~8.3f ms p99 ~8.3f ms~n", [Name, P(0.5), P(0.99)]). + +p50(L) -> + lists:nth(length(L) div 2 + 1, lists:sort(L)) / 1000. diff --git a/examples/bench_sessions_sdks.py b/examples/bench_sessions_sdks.py new file mode 100644 index 0000000..5dead01 --- /dev/null +++ b/examples/bench_sessions_sdks.py @@ -0,0 +1,98 @@ +"""What the isolation step costs in Temporal and Restate, for comparison with +examples/bench_sessions.erl. + +Temporal: SandboxedWorkflowRunner.prepare_workflow builds a workflow +sandbox, the way a new workflow run gets one (a fresh sys.modules in which +the workflow module is imported again; the standard library and pydantic +are passed through). + +Restate: the per-invocation state its Python SDK creates (the VM state +machine). Restate has no isolation between invocations of one process. + +Run with: + uv venv /tmp/sdk-venv && uv pip install -p /tmp/sdk-venv temporalio restate-sdk pydantic + /tmp/sdk-venv/bin/python examples/bench_sessions_sdks.py +""" + +import asyncio +import os +import statistics +import sys +import tempfile +import time + +WORKFLOW = ''' +from dataclasses import dataclass +import json, decimal, datetime +from temporalio import workflow +%s + +@dataclass +class Step: + n: int + +@workflow.defn +class Orders: + @workflow.run + async def run(self, state: dict) -> dict: + n = state.get('n', 0) + 1 + return {'commands': [['activity', 'charge', {'n': n}]], 'state': {'n': Step(n).n}} +''' + + +def pct(xs, q): + xs = sorted(xs) + return xs[max(0, min(len(xs) - 1, round(q * len(xs)) - 1))] + + +def row(name, xs, unit='ms', scale=1e3): + print(' %-48s p50 %9.3f %s p99 %9.3f %s' + % (name, pct(xs, .5) * scale, unit, pct(xs, .99) * scale, unit)) + + +def timed(fn, n): + fn() + out = [] + for _ in range(n): + t = time.perf_counter() + fn() + out.append(time.perf_counter() - t) + return out + + +PYDANTIC_MODEL = ''' +from pydantic import BaseModel + +class Order(BaseModel): + id: str + amount: float +''' + + +async def temporal(name, extra): + from temporalio.worker import UnsandboxedWorkflowRunner + from temporalio.worker.workflow_sandbox import SandboxedWorkflowRunner + from temporalio.workflow import _Definition + d = tempfile.mkdtemp() + with open(os.path.join(d, name + '.py'), 'w') as f: + f.write(WORKFLOW % extra) + sys.path.insert(0, d) + import importlib + mod = importlib.import_module(name) + defn = _Definition.must_from_class(mod.Orders) + label = 'with a pydantic model' if extra else 'stdlib only' + row('temporal sandbox per run, ' + label, timed(lambda: SandboxedWorkflowRunner().prepare_workflow(defn), 300)) + row('temporal unsandboxed, ' + label, timed(lambda: UnsandboxedWorkflowRunner().prepare_workflow(defn), 300)) + + +def restate(): + from restate.vm import VMWrapper + headers = [('content-type', 'application/vnd.restate.invocation.v5')] + row('restate: per-invocation state', timed(lambda: VMWrapper(headers), 5000), 'us', 1e6) + + +if __name__ == '__main__': + print('\n== isolation step per run / invocation ==') + asyncio.run(temporal('bench_orders_wf', '')) + asyncio.run(temporal('bench_orders_wf_pydantic', PYDANTIC_MODEL)) + restate() diff --git a/test/py_session_soak_SUITE.erl b/test/py_session_soak_SUITE.erl new file mode 100644 index 0000000..665436e --- /dev/null +++ b/test/py_session_soak_SUITE.erl @@ -0,0 +1,144 @@ +%%% @doc Soak test for sessions: many callers opening, using, crashing, +%%% killing and closing sessions from fork and spawn templates for a while, +%%% then resource counters checked against their baseline. It shows that +%%% every operation returns (no deadlock) and that nothing leaks: Erlang +%%% processes, ports, VM file descriptors, session children, zombies and +%%% scratch directories. +%%% +%%% Duration is 30 s by default; set `PY_SESSION_SOAK_SECONDS' to change it. +-module(py_session_soak_SUITE). + +-include_lib("common_test/include/ct.hrl"). + +-export([all/0, init_per_suite/1, end_per_suite/1]). +-export([test_session_churn_no_leak/1]). + +-define(MOD, py_test_session). + +all() -> [test_session_churn_no_leak]. + +init_per_suite(Config) -> + {ok, _} = application:ensure_all_started(erlang_python), + py:register_function(sess_reenter, fun([Ctx, N]) -> + {ok, R} = py_context:call(Ctx, ?MOD, reenter, [N - 1], #{}, 20000), + R + end), + [{test_dir, filename:join(code:lib_dir(erlang_python), "test")} | Config]. + +end_per_suite(_Config) -> + py:unregister_function(sess_reenter), + ok. + +test_session_churn_no_leak(Config) -> + Seconds = case os:getenv("PY_SESSION_SOAK_SECONDS") of + false -> 30; + S -> list_to_integer(S) + end, + TestDir = ?config(test_dir, Config), + {ok, Fork} = py_session:template(#{start => fork, zygotes => 2, paths => [TestDir], + imports => [?MOD]}), + {ok, Spawn} = py_session:template(#{start => spawn, warm => 2, paths => [TestDir], + imports => [?MOD]}), + timer:sleep(1000), + Base = counters(), + ct:print("baseline: ~p", [Base]), + Deadline = erlang:monotonic_time(millisecond) + Seconds * 1000, + Self = self(), + Workers = [spawn_link(fun() -> + rand:seed(exsss, {I, I * 7, I * 13}), + T = case I rem 4 of 0 -> Spawn; _ -> Fork end, + Self ! {done, self(), churn(T, Deadline, 0, [])} + end) || I <- lists:seq(1, 12)], + Results = [receive {done, W, R} -> R after (Seconds + 120) * 1000 -> ct:fail(worker_hung) end + || W <- Workers], + Ops = lists:sum([N || {N, _} <- Results]), + Errs = lists:append([E || {_, E} <- Results]), + ct:print("soak: ~p sessions in ~p s, ~p unexpected results: ~p", + [Ops, Seconds, length(Errs), lists:sublist(Errs, 5)]), + [] = Errs, + true = Ops > 0, + #{sessions := 0} = wait_idle(Fork, 100), + py_session:stop_template(Fork), + py_session:stop_template(Spawn), + timer:sleep(1000), + After = counters(), + ct:print("after: ~p", [After]), + check_no_growth(Base, After). + +%% One session per round, with a random use of it +churn(T, Deadline, N, Errs) -> + case erlang:monotonic_time(millisecond) < Deadline of + false -> + {N, Errs}; + true -> + {ok, S} = py_session:new(T), + Got = use(rand:uniform(8), S), + ok = py_session:close(S), + churn(T, Deadline, N + 1, case Got of ok -> Errs; Bad -> [Bad | Errs] end) + end. + +use(1, S) -> expect({ok, 3}, py_context:call(S, ?MOD, reenter, [3])); +use(2, S) -> expect({error, {child_exited, {signal, 6}}}, py_context:call(S, ?MOD, abort, [])); +use(3, S) -> expect(ok, py_context:kill(S)); +use(4, S) -> expect({error, timeout}, py_context:call(S, ?MOD, sleep, [10], #{}, 50)); +use(5, S) -> + %% a thread left running and a file written, then closed + _ = py_context:call(S, ?MOD, start_thread, []), + expect({ok, [<<"f">>]}, py_context:call(S, ?MOD, write_file, [<<"f">>, <<"x">>])); +use(6, S) -> expect({ok, none}, py_context:call(S, ?MOD, peek, [])); +use(7, S) -> expect(ok, py_context:exec(S, <<"import json; json.dumps(list(range(1000)))">>)); +use(8, _S) -> ok. + +expect(Want, Want) -> ok; +expect(Want, Got) -> {Want, Got}. + +wait_idle(T, 0) -> + py_session:info(T); +wait_idle(T, N) -> + case py_session:info(T) of + #{sessions := 0} = I -> I; + _ -> timer:sleep(50), wait_idle(T, N - 1) + end. + +%%% ============================================================================ +%%% Counters +%%% ============================================================================ + +counters() -> + erlang:garbage_collect(), + #{processes => erlang:system_info(process_count), + ports => erlang:system_info(port_count), + refs => ets:info(py_context_refs, size), + fds => beam_fd_count(), + children => session_children(), + zombies => zombies(), + scratch => length(filelib:wildcard(filename:join(py_child:sock_dir(), "sess_*")))}. + +check_no_growth(Base, After) -> + [begin + B = maps:get(K, Base), A = maps:get(K, After), + A =< B + Slack orelse ct:fail({leak, K, B, A}) + end || {K, Slack} <- [{processes, 5}, {ports, 0}, {refs, 0}, {children, 0}, + {zombies, 0}, {scratch, 0}, {fds, 8}]], + ok. + +beam_fd_count() -> + case os:type() of + {unix, linux} -> + length(filelib:wildcard("/proc/" ++ os:getpid() ++ "/fd/*")); + _ -> + Out = os:cmd("lsof -p " ++ os:getpid() ++ " 2>/dev/null | wc -l"), + list_to_integer(string:trim(Out)) - 1 + end. + +%% Python processes started by this node: spawned children, zygotes and the +%% sessions they forked (their command line is the zygote's) +session_children() -> + Out = os:cmd("ps -ax -o command= 2>/dev/null | grep -E 'py_zygote.py|py_isolated_child.py' " + "| grep erlang_python_" ++ os:getpid() ++ " | grep -v grep | wc -l"), + list_to_integer(string:trim(Out)). + +zombies() -> + Out = os:cmd("ps -ax -o stat=,command= 2>/dev/null | grep -E '^Z' " + "| grep -E 'python|py_zygote' | grep -v grep | wc -l"), + list_to_integer(string:trim(Out)). diff --git a/test/py_session_stress_SUITE.erl b/test/py_session_stress_SUITE.erl new file mode 100644 index 0000000..b8e27e6 --- /dev/null +++ b/test/py_session_stress_SUITE.erl @@ -0,0 +1,202 @@ +%%% @doc Stress and profiling for sessions. +%%% +%%% Numbers are logged (ct:print), not asserted tightly: they show what a +%%% fresh session costs next to a plain isolated context on the same +%%% machine. The asserts only catch regressions of an order of magnitude. +-module(py_session_stress_SUITE). + +-include_lib("common_test/include/ct.hrl"). + +-export([all/0, init_per_suite/1, end_per_suite/1, end_per_testcase/2]). + +-export([ + test_new_latency/1, + test_new_breakdown/1, + test_throughput/1, + test_call_overhead/1, + test_memory_per_session/1 +]). + +-define(MOD, py_test_session). + +all() -> [ + test_new_latency, + test_new_breakdown, + test_throughput, + test_call_overhead, + test_memory_per_session +]. + +init_per_suite(Config) -> + {ok, _} = application:ensure_all_started(erlang_python), + [{test_dir, filename:join(code:lib_dir(erlang_python), "test")} | Config]. + +end_per_suite(_Config) -> + ok. + +end_per_testcase(_Case, _Config) -> + [py_session:stop_template(T) || {_, T, _, _} <- supervisor:which_children(py_session_sup)], + ok. + +%% @doc Time to a usable session (new + first call), 100 in a row, for each +%% way of starting one, next to a plain isolated context. +test_new_latency(Config) -> + Fork = template(Config, #{start => fork}), + Spawn = template(Config, #{start => spawn}), + Warm = template(Config, #{start => spawn, warm => 4}), + N = 100, + Session = fun(T) -> + fun() -> + {Us, S} = us(fun() -> + {ok, S0} = py_session:new(T), + {ok, _} = py_context:call(S0, ?MOD, peek, []), + S0 + end), + py_session:close(S), + Us + end + end, + Plain = fun() -> + {Us, C} = us(fun() -> + {ok, C0} = py_context:new(#{mode => isolated, paths => [?config(test_dir, Config)]}), + {ok, _} = py_context:call(C0, ?MOD, peek, []), + C0 + end), + py_context:stop(C), + Us + end, + Rows = [{fork, run(Session(Fork), N)}, + {spawn, run(Session(Spawn), 20)}, + %% warm: paced so the pool refills between sessions + {warm_spawn, [begin timer:sleep(80), (Session(Warm))() end || _ <- lists:seq(1, 20)]}, + {plain_isolated, run(Plain, 20)}], + [ct:print("new+call ~-15s p50 ~7.2f ms p99 ~7.2f ms", [K, p(L, 50), p(L, 99)]) || {K, L} <- Rows], + ForkP50 = p(proplists:get_value(fork, Rows), 50), + SpawnP50 = p(proplists:get_value(spawn, Rows), 50), + true = ForkP50 < SpawnP50, + ok. + +%% @doc Where a forked session's time goes: the fork request, the child +%% connecting, and its `ready' frame. Same split as the prototype. +test_new_breakdown(Config) -> + T = template(Config, #{start => fork}), + Samples = [breakdown(T) || _ <- lists:seq(1, 100)], + [ct:print("fork breakdown ~-8s p50 ~6.3f ms p99 ~6.3f ms", + [K, p([maps:get(K, S) || S <- Samples], 50), p([maps:get(K, S) || S <- Samples], 99)]) + || K <- [fork, accept, ready, total]], + ok. + +breakdown(T) -> + Path = py_child:new_sock_path("bench_"), + {ok, L} = py_child:listen(Path), + T0 = erlang:monotonic_time(microsecond), + {ok, OsPid} = py_session_template:fork(T, Path, #{}, 5000), + T1 = erlang:monotonic_time(microsecond), + {ok, S} = py_child:accept(L, {os_pid, OsPid}, 5000), + T2 = erlang:monotonic_time(microsecond), + {ok, <<_:64/native, Len:32/native>>} = socket:recv(S, 12, 5000), + {ok, _Ready} = socket:recv(S, Len, 5000), + T3 = erlang:monotonic_time(microsecond), + socket:close(S), socket:close(L), file:delete(Path), + py_child:kill_os_pid(OsPid), + #{fork => T1 - T0, accept => T2 - T1, ready => T3 - T2, total => T3 - T0}. + +%% @doc Sessions per second (new + call + close) with 1, 8 and 64 callers, +%% for one and two zygotes. +test_throughput(Config) -> + Secs = 3, + Rows = [begin + T = template(Config, #{start => fork, zygotes => Z}), + R = [{C, throughput(T, C, Secs)} || C <- [1, 8, 64]], + py_session:stop_template(T), + {Z, R} + end || Z <- [1, 2]], + [[ct:print("zygotes ~w callers ~2w: ~7.1f sessions/s p50 ~6.2f ms p99 ~7.2f ms", + [Z, C, length(L) / Secs, p(L, 50), p(L, 99)]) || {C, L} <- R] || {Z, R} <- Rows], + ok. + +throughput(T, Callers, Secs) -> + Self = self(), + Deadline = erlang:monotonic_time(millisecond) + Secs * 1000, + Loop = fun L(Acc) -> + case erlang:monotonic_time(millisecond) < Deadline of + true -> + {Us, ok} = us(fun() -> + {ok, S} = py_session:new(T), + {ok, _} = py_context:call(S, ?MOD, peek, []), + py_session:close(S) + end), + L([Us | Acc]); + false -> + Acc + end + end, + Pids = [spawn_link(fun() -> Self ! {done, self(), Loop([])} end) || _ <- lists:seq(1, Callers)], + lists:append([receive {done, P, L} -> L after 60000 -> ct:fail(caller_hung) end || P <- Pids]). + +%% @doc A call in a session costs what it costs in an isolated context. +test_call_overhead(Config) -> + T = template(Config, #{start => fork}), + {ok, S} = py_session:new(T), + {ok, C} = py_context:new(#{mode => isolated, paths => [?config(test_dir, Config)]}), + Call = fun(Ctx) -> fun() -> element(1, us(fun() -> {ok, _} = py_context:call(Ctx, ?MOD, peek, []) end)) end end, + Reenter = fun(Ctx) -> fun() -> element(1, us(fun() -> {ok, 1} = py_context:call(Ctx, ?MOD, reenter, [1]) end)) end end, + py:register_function(sess_reenter, fun([Ctx, N]) -> + {ok, R} = py_context:call(Ctx, ?MOD, reenter, [N - 1]), R + end), + try + Rows = [{session_call, run(Call(S), 5000)}, + {isolated_call, run(Call(C), 5000)}, + {session_reenter, run(Reenter(S), 2000)}, + {isolated_reenter, run(Reenter(C), 2000)}], + [ct:print("~-17s p50 ~6.1f us p99 ~6.1f us", [K, p(L, 50) * 1000, p(L, 99) * 1000]) + || {K, L} <- Rows], + SessionP50 = p(proplists:get_value(session_call, Rows), 50), + PlainP50 = p(proplists:get_value(isolated_call, Rows), 50), + true = SessionP50 < PlainP50 * 2 + after + py:unregister_function(sess_reenter), + py_session:close(S), + py_context:stop(C) + end. + +%% @doc Resident memory of 20 live sessions, forked vs spawned. A forked +%% session shares the zygote's pages until it writes them. +test_memory_per_session(Config) -> + Rows = [begin + T = template(Config, #{start => Start}), + Ss = [begin {ok, S} = py_session:new(T), S end || _ <- lists:seq(1, 20)], + Pids = [begin {ok, P} = py_context:call(S, ?MOD, pid, []), P end || S <- Ss], + Rss = [rss_kb(P) || P <- Pids], + [py_session:close(S) || S <- Ss], + {Start, Rss} + end || Start <- [fork, spawn]], + [ct:print("~-5s rss per session p50 ~7.1f MB (ps rss counts shared pages in full)", + [K, lists:nth(length(L) div 2 + 1, lists:sort(L)) / 1024]) || {K, L} <- Rows], + ok. + +%%% ============================================================================ +%%% Helpers +%%% ============================================================================ + +template(Config, Opts) -> + {ok, T} = py_session:template(Opts#{paths => [?config(test_dir, Config)], + imports => [?MOD]}), + T. + +run(F, N) -> + _ = [F() || _ <- lists:seq(1, 3)], + [F() || _ <- lists:seq(1, N)]. + +us(F) -> + T0 = erlang:monotonic_time(microsecond), + R = F(), + {erlang:monotonic_time(microsecond) - T0, R}. + +%% Percentile in milliseconds +p(L, P) -> + S = lists:sort(L), + lists:nth(max(1, round(P / 100 * length(S))), S) / 1000. + +rss_kb(OsPid) -> + list_to_integer(string:trim(os:cmd("ps -o rss= -p " ++ integer_to_list(OsPid)))). From efdfaa7e1c28f64f69b2e92322dcdd3068c71706 Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Sat, 26 Sep 2026 11:04:28 +0200 Subject: [PATCH 06/18] Document isolated sessions A task-oriented guide (template, new, run, fork or spawn, what a session sees, limits, costs), the decision record for forking sessions from a zygote, the exited state of a session context, and the changelog. --- CHANGELOG.md | 25 +++- docs/decisions/0009-isolated-sessions.md | 52 +++++++ docs/decisions/overview.md | 1 + docs/isolated.md | 3 + docs/security.md | 4 +- docs/sessions.md | 172 +++++++++++++++++++++++ docs/state-machines.md | 11 +- rebar.config | 6 +- test/coverage_audit.md | 4 + 9 files changed, 274 insertions(+), 4 deletions(-) create mode 100644 docs/decisions/0009-isolated-sessions.md create mode 100644 docs/sessions.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 8b1ad49..44871f8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,6 +1,29 @@ # Changelog -## 5.0.1 (2026-09-22) +## 5.1.0 (unreleased) + +### Added + +- **Isolated sessions** - `py_session:template/1` prepares a Python + environment once (interpreter, `paths`, `imports`, `preload`, `env`, + `hash_seed`, `rlimits`) and `py_session:new/1` gives a fresh child process + per session that starts from it and shares no state with any other + session: module globals, `sys.modules`, environment, threads, working + directory. By default a session is forked from a zygote that already ran + the imports and preload, so it is ready in a few milliseconds instead of + the ~50 ms of a new interpreter plus its imports; `start => spawn` starts + a new interpreter per session (optionally from a `warm` pool) for code + that cannot be forked. A session is an isolated context: calls, callbacks, + calls back into the same session, interrupts, loops and `pass_fd` work + unchanged. `close/1` kills it; a session whose process dies answers with + the reason until closed and is never restarted. `run/5` runs one call in a + new session; `refresh/1` rebuilds the template after a deploy; `info/1` + reports zygotes, live sessions and forks. See `docs/sessions.md`. +- `clear_env` and `hash_seed` options for isolated contexts: the child sees + only the variables named in `env`, and every child built with the same + seed hashes strings and orders sets the same way. + + ### Fixed diff --git a/docs/decisions/0009-isolated-sessions.md b/docs/decisions/0009-isolated-sessions.md new file mode 100644 index 0000000..13fd9c3 --- /dev/null +++ b/docs/decisions/0009-isolated-sessions.md @@ -0,0 +1,52 @@ +# 0009: Sessions fork from a prepared zygote + +Since 5.1.0. Code: `src/py_session.erl`, `src/py_session_template.erl`, +`priv/py_zygote.py`, `src/py_isolated.erl` (session origin), `src/py_child.erl`. + +## Situation + +Callers such as a durable-execution engine need each run of a Python +function to start from the same state and to leave nothing behind for the +next run. An isolated context keeps its interpreter between calls, and a +new one costs a cold interpreter start plus every import (about 50 ms +before the first import). Temporal's Python SDK isolates a workflow run by +re-importing its module in a fresh `sys.modules` inside a shared process; +Restate's does not isolate invocations at all and relies on the platform +(a container, a Lambda microVM). Neither gives a fresh process per run. + +## Decision + +A template prepares an interpreter once, in a zygote: a single-threaded +child that runs the imports and preload, builds the child runtime (the +`_isolated.Runtime` and the `erlang` module) without a socket, then forks +one child per session. The forked child connects that runtime to its own +socket and continues as a normal isolated child, so a session is an +isolated context (`py_isolated` with `session => true`). The zygote reports +each child's exit on its control socket; the template forwards it to the +session's context. `start => spawn` starts a normal child per session +instead, with an optional warm pool, for templates that cannot be forked. + +A session is never restarted: after its child dies it answers every +request with the reason until it is closed. It runs in its own scratch +directory, its stdio is detached from the zygote's port, and it sees only +the template's environment. + +Not chosen: a subinterpreter per session (about 13 ms, shares the process +environment, working directory, hash seed and C-extension state, and PyO3 +extensions refuse to load), Temporal-style re-import in one process (state +leaks through passthrough and C modules), CRIU (Linux only, needs +privileges and PID namespaces), and Wasm images (no native C extensions). + +## Consequences + +- A session costs a fork and a connect (a few milliseconds) instead of an + interpreter start and the imports. +- The zygote must stay single threaded: a template whose imports start a + thread is refused. On macOS, modules that load Objective-C cannot be + forked safely; `start => spawn` is the way out. +- All sessions of a template share its hash seed and its prepared state; + per-session randomness relies on Python's at-fork reseeding. +- `erlang` functions fail during preload: the runtime is not connected + until a session exists. +- A zygote that dies is rebuilt; its orphaned sessions are watched by pid + (`py_isolated` probes `kill(pid, 0)`), since nobody reports their exit. diff --git a/docs/decisions/overview.md b/docs/decisions/overview.md index 0d1f7eb..a6e2e3a 100644 --- a/docs/decisions/overview.md +++ b/docs/decisions/overview.md @@ -16,3 +16,4 @@ what was decided, what it costs, and where the code is. | [0006](0006-shared-memory-over-iommap.md) | Bulk data through iommap regions, handles as plain tuples | 5.0.0 | | [0007](0007-remove-legacy-execution-paths.md) | One execution path per mode; the legacy API is removed | 5.0.0 | | [0008](0008-pipe-io-rules.md) | Pipe I/O is non-blocking, deadlined and waited with poll | 3.1.0, 5.0.0 | +| [0009](0009-isolated-sessions.md) | Sessions fork from a prepared zygote | 5.1.0 | diff --git a/docs/isolated.md b/docs/isolated.md index 067be8c..009d517 100644 --- a/docs/isolated.md +++ b/docs/isolated.md @@ -24,6 +24,9 @@ memory bound. The public API is the one you already use with `worker` and | Startup | microseconds | milliseconds | ~40 ms | | Zero-copy `py_buffer`, channels, `erlang.schedule`, object refs | yes | yes | no (see Limits) | +To give every request or job its own fresh child, started from prepared +imports in a few milliseconds, use [Isolated Sessions](sessions.md). + ## Start a context ```erlang diff --git a/docs/security.md b/docs/security.md index 858b718..a2af68a 100644 --- a/docs/security.md +++ b/docs/security.md @@ -158,7 +158,9 @@ child process: ``` A crash kills only the child, `py_context:kill/1` is total, and rlimits or -cgroups bound resources. See [Isolated Contexts](isolated.md). +cgroups bound resources. See [Isolated Contexts](isolated.md). When each +request must also start from a clean state, with nothing left by the +previous one, give it its own session: see [Isolated Sessions](sessions.md). That is a boundary against Python *failing*, not against Python *reaching*. The child runs as the same user as the node, so it can read and write diff --git a/docs/sessions.md b/docs/sessions.md new file mode 100644 index 0000000..2064813 --- /dev/null +++ b/docs/sessions.md @@ -0,0 +1,172 @@ +# Isolated Sessions + +A session is a fresh Python process for one piece of work: a request, a +workflow step, a user's job. You prepare a template once (interpreter, +paths, imports, preload code, environment, limits), then every +`py_session:new/1` gives you a new process that starts from that template +and shares no state with any other session. Use sessions when one run must +never see what another run left behind: module globals, `sys.modules`, +environment changes, threads, files in its working directory. + +## Build a template + +```erlang +{ok, T} = py_session:template(#{ + paths => ["/srv/app"], + imports => [orders], + preload => <<"import orders\norders.load_rules()">>, + env => #{"TZ" => "UTC"}, + hash_seed => 0 +}). +``` + +The imports and preload run once, now. With the default `start => fork` +they run in a zygote process that every session is forked from, so a +session starts with them already done. + +## Open a session and use it + +```erlang +{ok, S} = py_session:new(T), +{ok, Result} = py_context:call(S, orders, handle, [Event, State]), +ok = py_session:close(S). +``` + +A session is an isolated context: `py_context:call/eval/exec`, callbacks +with `erlang.call`, calls back into the same session from a callback, +`py_context:interrupt/1`, `kill/1`, worker loops and `pass_fd/2` all work +on it. It is linked to the process that created it. `close/1` kills its +process; a session is never reused. + +## Run one call + +```erlang +{ok, Result} = py_session:run(T, orders, handle, [Event, State], + #{timeout => 5000}). +``` + +`run/5` opens a session, makes the call and closes it, also when the call +fails. + +## Choose how sessions start + +| `start` | How a session starts | Use it when | +|---|---|---| +| `fork` (default) | forked from the template's zygote: imports and preload are already done | the template can be forked | +| `spawn` | a new interpreter runs the imports and preload | a module starts a thread when imported, or loads Objective-C on macOS | + +A fork copies only the thread that calls it, so a template whose imports +start a thread is refused: + +```erlang +{error, {template_failed, {init_failed, {threads, [<<"import-time-thread">>]}}}} = + py_session:template(#{imports => [module_with_a_thread]}). +``` + +With `start => spawn`, keep sessions started ahead of time so `new/1` does +not wait for an interpreter: + +```erlang +{ok, T} = py_session:template(#{start => spawn, warm => 4, imports => [orders]}). +``` + +With `start => fork`, `zygotes => N` runs N zygotes that fork in turn when +one is not enough. + +## What a session sees + +| | Sessions of one template | +|---|---| +| Module globals, `sys.modules`, `__main__` | fresh per session | +| `os.environ` | the template's `env` only; nothing from the VM unless `clear_env => false` | +| Working directory | an empty directory per session, removed on close | +| Threads, open files, sockets | none from another session | +| `random` state | reseeded per session | +| Hash seed (`set` and `dict` of `str` order) | the same for every session: `hash_seed`, or one chosen at build time | +| Imports and preload globals | the same starting state for every session | + +`print` in a session goes to the Erlang logger, tagged with the session's +context. + +## Bound each session + +```erlang +{ok, T} = py_session:template(#{ + imports => [orders], + rlimits => #{as => 512 * 1024 * 1024, cpu => 10, nofile => 256}, + kill_after => 1000 +}). +``` + +The limits apply to every session, as for isolated contexts (see +[Isolated Contexts](isolated.md)). A timeout on a call interrupts it; +`py_context:kill/1` ends the session's process at once. + +## When a session dies + +```erlang +{error, {child_exited, {signal, 6}}} = py_context:call(S, orders, crash, []), +{error, {child_exited, {signal, 6}}} = py_context:eval(S, <<"1">>), +ok = py_session:close(S). +``` + +A session whose process dies is not restarted: every request answers with +the reason until you close it. Other sessions and the template are not +affected. If a zygote dies, the template starts a new one; the sessions it +had forked keep running. + +## Refresh after a deploy + +```erlang +ok = py_session:refresh(T). +``` + +New sessions start from the new code; sessions already open keep what they +started with. + +## Inspect a template + +```erlang +#{start := fork, zygotes := [#{os_pid := _}], sessions := Live, forks := Total} = + py_session:info(T). +``` + +## What it costs + +Measured with `examples/bench_sessions.erl`, new + first call + close, +p50. Run it on your machine for your own figures: the cost of a fork grows +with the size of the prepared process. + +| How the session starts | macOS 27, Python 3.14 | Linux (container), Python 3.11 | +|---|---|---| +| `fork` | ~4 ms | ~4 ms | +| `spawn` with a `warm` pool that keeps up | ~2 ms | ~2 ms | +| `spawn` | ~60 ms | ~45 ms | +| a plain isolated context, for reference | ~60 ms | ~40 ms | + +One zygote forks one session at a time. On Linux it served about 1,000 +sessions a second with 8 or more callers; add `zygotes` when sessions are +opened faster than one zygote forks them. + +A call inside a session costs what it costs in any isolated context, since +it crosses the same socket. `examples/bench_sessions_sdks.py` measures the +isolation step of Temporal's workflow sandbox and Restate's SDK on the same +workflow module. + +## Limits + +- A session is a process boundary, not a security boundary: it runs as the + node's user and can reach what that user can. See [Security](security.md). +- `erlang.call` and the other `erlang` functions work in a session, not + while the template prepares: preload code that calls Erlang fails with + "erlang is not connected". +- On macOS a module that loads Objective-C frameworks cannot be forked + safely; use `start => spawn` for it. +- All sessions of a template share one hash seed, so they order sets the + same way. `refresh/1` picks a new one when `hash_seed` is not set. + +## See also + +- [Isolated Contexts](isolated.md): the options sessions share +- [Security](security.md): what a process boundary does and does not bound +- [Decision 0009](decisions/0009-isolated-sessions.md): why sessions fork from a zygote diff --git a/docs/state-machines.md b/docs/state-machines.md index f2fee76..928531f 100644 --- a/docs/state-machines.md +++ b/docs/state-machines.md @@ -101,6 +101,8 @@ shows the current one and `sys:trace/2` prints transitions. +-- child exit / kill / socket error --> {restarting, Reason} --new child--> idle | +-- budget exhausted --> stop + | + +-- session => true --> {exited, Reason} ``` Per state: @@ -111,7 +113,8 @@ Per state: | `{busy, Id}` | postponed, unless from a process running a callback for this context (nested, dispatched) | dispatched | `{timeout, kill}` bound to `Id` once an interrupt was sent | | `looping` | `{error, loop_running}` | dispatched (`submit`, `pass_fd`, ...) | none | | `stopping_loop` | postponed | dispatched | `state_timeout` for the interrupt, then `{timeout, kill}` bound to `loop` | -| `{restarting, R}` | postponed | postponed | `state_timeout` waiting for the port's `exit_status` | +| `{restarting, R}` | postponed | postponed | `state_timeout` waiting for the port's `exit_status`; for a forked session child, a probe of its pid every 100 ms | +| `{exited, R}` | `{error, R}` | `{error, R}` | none | Transitions and their triggers: @@ -126,6 +129,12 @@ Transitions and their triggers: passed the handshake. `restart_allowed/1` counts restarts in `restart_period`; over `max_restarts` (or with `restart => false`) the process stops with `{child_exited, Reason}`. +- `{restarting, _}` to `{exited, Reason}`: a session (`session => true`, + started by `py_session`) is never restarted. Its child's exit comes from + the port, or for a child forked by a template from the template's + `{py_session_exited, OsPid, Code}`, or from the pid probe when the + template is gone. The session answers every request with `Reason` and + `kill/1` callers are answered at once; `stop` ends it. - `looping` to `stopping_loop`: `stop_loop/2` or the owner's `DOWN`. `stopping_loop` to `idle`: the `{loop_exit, R}` event. The interrupt `state_timeout` and the kill backstop escalate if the loop does not exit. diff --git a/rebar.config b/rebar.config index ddae020..898a66d 100644 --- a/rebar.config +++ b/rebar.config @@ -71,6 +71,7 @@ <<"docs/asyncio.md">>, <<"docs/workers.md">>, <<"docs/isolated.md">>, + <<"docs/sessions.md">>, <<"docs/reactor.md">>, <<"docs/process-bound-envs.md">>, <<"docs/security.md">>, @@ -91,6 +92,7 @@ <<"docs/decisions/0006-shared-memory-over-iommap.md">>, <<"docs/decisions/0007-remove-legacy-execution-paths.md">>, <<"docs/decisions/0008-pipe-io-rules.md">>, + <<"docs/decisions/0009-isolated-sessions.md">>, <<"docs/preload.md">>, <<"docs/owngil_internals.md">>, <<"docs/event_loop_architecture.md">> @@ -119,6 +121,7 @@ <<"docs/asyncio.md">>, <<"docs/workers.md">>, <<"docs/isolated.md">>, + <<"docs/sessions.md">>, <<"docs/reactor.md">>, <<"docs/process-bound-envs.md">>, <<"docs/security.md">>, @@ -145,7 +148,8 @@ <<"docs/decisions/0005-py-isolated-gen-statem.md">>, <<"docs/decisions/0006-shared-memory-over-iommap.md">>, <<"docs/decisions/0007-remove-legacy-execution-paths.md">>, - <<"docs/decisions/0008-pipe-io-rules.md">> + <<"docs/decisions/0008-pipe-io-rules.md">>, + <<"docs/decisions/0009-isolated-sessions.md">> ]} ]} ]}. diff --git a/test/coverage_audit.md b/test/coverage_audit.md index bb8a8ca..25076ee 100644 --- a/test/coverage_audit.md +++ b/test/coverage_audit.md @@ -19,6 +19,10 @@ suite is visible. `scripts/check_code_map.sh` requires a row per module. | `py_venv` | `py_venv_SUITE` | | `py_shared_dict` | `py_SUITE` | | `py_isolated` | `py_isolated_*_SUITE` | +| `py_child` | `py_isolated_*_SUITE`, `py_session_SUITE` | +| `py_session` | `py_session_SUITE`, `py_session_stress_SUITE`, `py_session_soak_SUITE` | +| `py_session_template` | `py_session_SUITE`, `py_session_stress_SUITE`, `py_session_soak_SUITE` | +| `py_session_sup` | (through `py_session_SUITE`) | | `py_context_router` | `py_context_router_SUITE`, `py_pool_SUITE` | | `py_context_sup` | (through the above) | | `py_context_init` | (through the above) | From 3ec949dff3fdc3a359a609f70264d150dfe0b979 Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Sat, 26 Sep 2026 11:14:30 +0200 Subject: [PATCH 07/18] Keep session start and close off the file server Every file:make_dir, del_dir_r, delete and change_mode call goes through the node's single file server, so sessions opened in parallel queued behind each other: on macOS eight callers got fewer sessions a second than one. The socket directory is now created once and cached, and a session's socket file and scratch directory are handled with prim_file from its own process. close/1 replies before the directory is removed. --- src/py_child.erl | 79 ++++++++++++++++++++++++++++++++----- src/py_isolated.erl | 22 +++++------ src/py_session_template.erl | 2 +- test/py_session_SUITE.erl | 11 +++++- 4 files changed, 90 insertions(+), 24 deletions(-) diff --git a/src/py_child.erl b/src/py_child.erl index 62ca7d5..7b8eba4 100644 --- a/src/py_child.erl +++ b/src/py_child.erl @@ -29,6 +29,9 @@ priv_dir/0, sock_dir/0, new_sock_path/1, + delete_file/1, + make_scratch_dir/1, + remove_tree/1, listen/1, accept/3, tune_socket/1, @@ -46,6 +49,8 @@ to_bin/1, to_list/1]). +-include_lib("kernel/include/file.hrl"). + -define(SOCKET_BUF, 1024 * 1024). %% @doc Python executable used for children: the `python' option, then the @@ -100,17 +105,71 @@ priv_dir() -> filename:absname(Dir). %% @doc Private directory for the sockets of this node (mode 0700). Kept -%% under `$TMPDIR': a Unix socket path is limited to 104 bytes. +%% under `$TMPDIR': a Unix socket path is limited to 104 bytes. Created once; +%% the path is cached so opening a child does not go through the file server. -spec sock_dir() -> file:filename(). sock_dir() -> - Base = case os:getenv("TMPDIR") of - false -> "/tmp"; - T -> T - end, - Dir = filename:join(Base, "erlang_python_" ++ os:getpid()), - ok = filelib:ensure_dir(filename:join(Dir, "x")), - _ = file:change_mode(Dir, 8#700), - Dir. + case persistent_term:get({?MODULE, sock_dir}, undefined) of + undefined -> + Base = case os:getenv("TMPDIR") of + false -> "/tmp"; + T -> T + end, + Dir = filename:join(Base, "erlang_python_" ++ os:getpid()), + ok = filelib:ensure_dir(filename:join(Dir, "x")), + _ = file:change_mode(Dir, 8#700), + persistent_term:put({?MODULE, sock_dir}, Dir), + Dir; + Dir -> + Dir + end. + +%% Files of a child (socket paths, session directories) are handled with +%% prim_file, as `raw' file access is: from the calling process, not +%% through the node's single file server, which would serialise every +%% session start and close. + +%% @doc Delete a file, ignoring errors. +-spec delete_file(file:filename()) -> ok. +delete_file(Path) -> + _ = prim_file:delete(Path), + ok. + +%% @doc A new empty directory under sock_dir/0. +-spec make_scratch_dir(string()) -> {ok, file:filename()} | {error, term()}. +make_scratch_dir(Prefix) -> + Dir = filename:join(sock_dir(), Prefix ++ integer_to_list(erlang:unique_integer([positive]))), + case prim_file:make_dir(Dir) of + ok -> + {ok, Dir}; + {error, enoent} -> + %% the socket directory was removed (a tmp cleaner): make it again + persistent_term:erase({?MODULE, sock_dir}), + _ = sock_dir(), + case prim_file:make_dir(Dir) of + ok -> {ok, Dir}; + {error, _} = Err -> Err + end; + {error, _} = Err -> + Err + end. + +%% @doc Remove a directory and everything in it, ignoring errors. +-spec remove_tree(file:filename()) -> ok. +remove_tree(Path) -> + case prim_file:read_link_info(Path) of + {ok, #file_info{type = directory}} -> + case prim_file:list_dir(Path) of + {ok, Names} -> [remove_tree(filename:join(Path, N)) || N <- Names]; + _ -> ok + end, + _ = prim_file:del_dir(Path), + ok; + {ok, _} -> + delete_file(Path); + {error, _} -> + ok + end. -spec new_sock_path(string()) -> file:filename(). new_sock_path(Prefix) -> @@ -120,7 +179,7 @@ new_sock_path(Prefix) -> %% @doc Listening socket at Path, for one child to connect to. -spec listen(file:filename()) -> {ok, socket:socket()} | {error, term()}. listen(Path) -> - _ = file:delete(Path), + delete_file(Path), case socket:open(local, stream, default) of {ok, L} -> case socket:bind(L, #{family => local, path => Path}) of diff --git a/src/py_isolated.erl b/src/py_isolated.erl index 493bcab..3797eda 100644 --- a/src/py_isolated.erl +++ b/src/py_isolated.erl @@ -353,8 +353,8 @@ handle_event(info, {stop, From, MRef}, _State, #data{opts = Opts} = Data) -> false -> graceful end, Data1 = stop_child(Data, How), - %% Gone before the caller hears back: a closed session leaves nothing - remove_scratch(Data1), + %% The scratch directory is removed by terminate/3, right after the + %% reply: the caller does not wait for the file system From ! {MRef, ok}, {stop, normal, Data1}; @@ -520,9 +520,7 @@ start_child_1(#data{opts = Opts} = St0) -> %% A session runs in its own empty directory, created here and removed %% when the context stops. with_scratch(#data{scratch = undefined, opts = #{session := true}} = St) -> - Dir = filename:join(py_child:sock_dir(), - "sess_" ++ integer_to_list(erlang:unique_integer([positive]))), - ok = file:make_dir(Dir), + {ok, Dir} = py_child:make_scratch_dir("sess_"), St#data{scratch = Dir}; with_scratch(St) -> St. @@ -530,7 +528,7 @@ with_scratch(St) -> remove_scratch(#data{scratch = undefined}) -> ok; remove_scratch(#data{scratch = Dir}) -> - _ = file:del_dir_r(Dir), + py_child:remove_tree(Dir), ok. %% Ask the session template to fork a child from its zygote; the child @@ -545,18 +543,18 @@ fork_child(Template, Opts) -> {ok, OsPid} -> case py_child:accept(L, {os_pid, OsPid}, Timeout) of {ok, S} -> - _ = file:delete(Path), + py_child:delete_file(Path), py_child:tune_socket(S), {ok, #child{port = undefined, os_pid = OsPid, listener = L, sock = S, sock_path = Path}}; {error, Reason} -> - _ = file:delete(Path), + py_child:delete_file(Path), socket:close(L), py_child:kill_os_pid(OsPid), {error, Reason} end; {error, Reason} -> - _ = file:delete(Path), + py_child:delete_file(Path), socket:close(L), {error, Reason} end; @@ -583,19 +581,19 @@ spawn_child(Python, Opts) -> Timeout = maps:get(start_timeout, Opts, ?DEFAULT_START_TIMEOUT_MS), case py_child:accept(L, {port, Port}, Timeout) of {ok, S} -> - _ = file:delete(Path), + py_child:delete_file(Path), py_child:tune_socket(S), {ok, #child{port = Port, os_pid = OsPid, listener = L, sock = S, sock_path = Path}}; {error, Reason} -> - _ = file:delete(Path), + py_child:delete_file(Path), socket:close(L), kill_port(Port, OsPid), {error, Reason} end catch Class:Err:Stack -> - _ = file:delete(Path), + py_child:delete_file(Path), socket:close(L), {error, {spawn_failed, {Class, Err, Stack}}} end; diff --git a/src/py_session_template.erl b/src/py_session_template.erl index e03b10b..0c54cdc 100644 --- a/src/py_session_template.erl +++ b/src/py_session_template.erl @@ -303,7 +303,7 @@ start_zygote(Opts) -> {error, _} = Err -> Err end, - _ = file:delete(Path), + py_child:delete_file(Path), case Result of {ok, _} = Ok -> Ok; diff --git a/test/py_session_SUITE.erl b/test/py_session_SUITE.erl index 0b96ca0..07f0262 100644 --- a/test/py_session_SUITE.erl +++ b/test/py_session_SUITE.erl @@ -313,9 +313,18 @@ test_close_reaps_and_cleans(Config) -> true = filelib:is_dir(Cwd), ok = py_session:close(S), ok = wait_gone(Pid), - false = filelib:is_dir(Cwd), + ok = wait_no_dir(Cwd, 100), ok. +%% removed by the context process right after close/1 returns +wait_no_dir(Dir, 0) -> + ct:fail({still_there, Dir}); +wait_no_dir(Dir, N) -> + case filelib:is_dir(Dir) of + false -> ok; + true -> timer:sleep(20), wait_no_dir(Dir, N - 1) + end. + %% A session is linked to the process that created it. test_caller_crash_stops_session(Config) -> T = template(Config), From 50818b70fb0b3f5dd0ba36c075ecb83d41c17300 Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Sat, 26 Sep 2026 11:32:04 +0200 Subject: [PATCH 08/18] Serve nested calls on an owngil context while erlang.call waits An owngil context thread waited for erlang.call on its callback pipe, so a callback calling back into the same context queued behind the request that was waiting for it, until the timeout. Requests with a caller now wait inline and serve the nested call, as on a worker context. --- CHANGELOG.md | 8 +++++ c_src/py_callback.c | 7 +++- c_src/py_nif.c | 8 +++++ test/py_reentrant_SUITE.erl | 68 +++++++++++++++++++++++++++++++++++-- 4 files changed, 88 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 44871f8..126d67a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -23,6 +23,14 @@ only the variables named in `env`, and every child built with the same seed hashes strings and orders sets the same way. +### Fixed + +- On an owngil context, a Python function that called `erlang.call`, where + the Erlang callback called the same context again, hung until the request + timeout. The context thread waited for the callback on its pipe while the + nested call sat in its queue. It now waits inline and serves the nested + call, as worker contexts do since 5.0.1. + ### Fixed diff --git a/c_src/py_callback.c b/c_src/py_callback.c index 6ac4126..4522483 100644 --- a/c_src/py_callback.c +++ b/c_src/py_callback.c @@ -1657,9 +1657,14 @@ static PyObject *erlang_call_impl(PyObject *self, PyObject *args) { bool has_context_suspension = (tl_current_context != NULL && tl_allow_suspension && !loop_running); bool has_context_handler = (tl_current_context != NULL && tl_current_context->has_callback_handler); + /* An owngil context also has a callback handler (for its other + * threads), but a request with a caller must wait inline too: on the + * handler pipe a callback calling back into this context would queue + * behind the request that waits for it. */ bool has_context_inline = (tl_current_context != NULL && !tl_allow_suspension && tl_current_context->has_current_caller && - !has_context_handler && !loop_running); + (!has_context_handler || tl_current_context->is_subinterp) && + !loop_running); if (has_context_inline) { Py_ssize_t nargs = PyTuple_Size(args); diff --git a/c_src/py_nif.c b/c_src/py_nif.c index 54d77d6..7ff4fa5 100644 --- a/c_src/py_nif.c +++ b/c_src/py_nif.c @@ -3166,6 +3166,13 @@ static void *ctx_thread_main_owngil(void *arg) { ctx->request_term = req->request_data; ctx->reactor_buffer_ptr = req->reactor_buffer_ptr; ctx->local_env_ptr = req->local_env_ptr; + /* The caller serves erlang.call made by this request: the call + * waits inline and serves nested requests (ctx_call_erlang_inline), + * as on a worker context thread */ + ctx->has_current_caller = req->async_mode; + if (req->async_mode) { + ctx->current_caller = req->caller_pid; + } ctx->response_ok = false; ctx->response_term = 0; @@ -3195,6 +3202,7 @@ static void *ctx_thread_main_owngil(void *arg) { ctx->request_term = 0; ctx->reactor_buffer_ptr = NULL; ctx->local_env_ptr = NULL; + ctx->has_current_caller = false; /* Deliver result - async (message to caller) or blocking (condvar) */ if (req->async_mode) { diff --git a/test/py_reentrant_SUITE.erl b/test/py_reentrant_SUITE.erl index 47af7be..599faf1 100644 --- a/test/py_reentrant_SUITE.erl +++ b/test/py_reentrant_SUITE.erl @@ -31,7 +31,9 @@ test_call_reentrant_depths/1, test_call_reentrant_repeated/1, test_call_sequential_callbacks/1, - test_readme_reentrant_example/1 + test_readme_reentrant_example/1, + test_owngil_call_reentrant_depths/1, + test_owngil_concurrent_reentrant/1 ]). all() -> @@ -52,7 +54,9 @@ all() -> test_call_reentrant_depths, test_call_reentrant_repeated, test_call_sequential_callbacks, - test_readme_reentrant_example + test_readme_reentrant_example, + test_owngil_call_reentrant_depths, + test_owngil_concurrent_reentrant ]. init_per_suite(Config) -> @@ -83,6 +87,7 @@ end_per_testcase(_TestCase, _Config) -> try py:unregister_function(etf_probe_novel) catch _:_ -> ok end, try py:unregister_function(rs_double) catch _:_ -> ok end, try py:unregister_function(nest_step) catch _:_ -> ok end, + try py:unregister_function(owngil_nest) catch _:_ -> ok end, ok. %%% ============================================================================ @@ -115,6 +120,65 @@ test_call_reentrant_depths(_Config) -> py:unregister_function(nest_step), ok. +%%% owngil: the context thread used to block on the callback pipe, so a +%%% callback calling back into the same owngil context queued behind the +%%% request waiting for it and the chain hung until the timeout. + +-define(OWNGIL_DOWN, <<" +import erlang +def odown(ctx, n, d): + if d == 0: + return n + return erlang.call('owngil_nest', ctx, n, d) +">>). + +owngil_ctx() -> + {ok, C} = py_context:new(#{mode => owngil}), + ok = py_context:exec(C, ?OWNGIL_DOWN), + C. + +register_owngil_nest() -> + py:register_function(owngil_nest, fun([Ctx, N, D]) -> + {ok, R} = py_context:call(Ctx, '__main__', odown, [Ctx, N + 1, D - 1], #{}, 10000), + R + end). + +%% @doc py_context:call on an owngil context -> erlang.call -> the same +%% context, 1 to 5 deep, then repeated. +test_owngil_call_reentrant_depths(_Config) -> + case py_nif:owngil_supported() of + false -> + {skip, "OWN_GIL requires Python 3.14+"}; + true -> + register_owngil_nest(), + C = owngil_ctx(), + [{ok, D} = py_context:call(C, '__main__', odown, [C, 0, D], #{}, 10000) + || D <- [1, 2, 3, 5]], + [{ok, 2} = py_context:call(C, '__main__', odown, [C, 0, 2], #{}, 10000) + || _ <- lists:seq(1, 20)], + py_context:stop(C) + end. + +%% @doc Several owngil contexts nesting at once, each into itself. +test_owngil_concurrent_reentrant(_Config) -> + case py_nif:owngil_supported() of + false -> + {skip, "OWN_GIL requires Python 3.14+"}; + true -> + register_owngil_nest(), + Ctxs = [owngil_ctx() || _ <- lists:seq(1, 4)], + Self = self(), + Pids = [spawn_link(fun() -> + C = lists:nth(I rem 4 + 1, Ctxs), + Self ! {done, [py_context:call(C, '__main__', odown, [C, 0, 3], #{}, 10000) + || _ <- lists:seq(1, 25)]} + end) || I <- lists:seq(1, 8)], + [receive {done, R} -> R = lists:duplicate(25, {ok, 3}) after 60000 -> ct:fail(timeout) end + || _ <- Pids], + [py_context:stop(C) || C <- Ctxs], + ok + end. + %% @doc One depth, many times: the nested py:call must not depend on which %% context the scheduler picks. test_call_reentrant_repeated(_Config) -> From 340f0c98f82612c5a837065509d7badadf1bbf4e Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Sat, 26 Sep 2026 11:51:12 +0200 Subject: [PATCH 09/18] Look the erlang module up in the interpreter's table The NIF read sys.modules to find the erlang module when it extends it. When sys.modules is replaced by a mapping (a re-import template does that), PyDict_GetItemString on it returns nothing and the extension code ran without erlang. PyImport_GetModuleDict() is the table the import system itself uses. --- c_src/py_callback.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/c_src/py_callback.c b/c_src/py_callback.c index 4522483..7b0d1bb 100644 --- a/c_src/py_callback.c +++ b/c_src/py_callback.c @@ -3799,7 +3799,7 @@ static int create_erlang_module(void) { PyDict_SetItemString(log_globals, "__builtins__", builtins); /* Import erlang module into globals so the code can reference it */ - PyObject *sys_modules = PySys_GetObject("modules"); + PyObject *sys_modules = PyImport_GetModuleDict(); /* the interpreter's table, a dict even when sys.modules is replaced */ if (sys_modules != NULL) { PyObject *erlang_mod = PyDict_GetItemString(sys_modules, "erlang"); if (erlang_mod != NULL) { @@ -3901,7 +3901,7 @@ static int create_erlang_module(void) { PyDict_SetItemString(ext_globals, "__builtins__", builtins); /* Import erlang module into globals so the code can reference it */ - PyObject *sys_modules = PySys_GetObject("modules"); + PyObject *sys_modules = PyImport_GetModuleDict(); /* the interpreter's table, a dict even when sys.modules is replaced */ if (sys_modules != NULL) { PyObject *erlang_mod = PyDict_GetItemString(sys_modules, "erlang"); if (erlang_mod != NULL) { @@ -3958,7 +3958,7 @@ static int create_erlang_module(void) { PyDict_SetItemString(atom_globals, "__builtins__", builtins); /* Import erlang module into globals so the code can reference it */ - PyObject *sys_modules = PySys_GetObject("modules"); + PyObject *sys_modules = PyImport_GetModuleDict(); /* the interpreter's table, a dict even when sys.modules is replaced */ if (sys_modules != NULL) { PyObject *erlang_mod = PyDict_GetItemString(sys_modules, "erlang"); if (erlang_mod != NULL) { @@ -4048,7 +4048,7 @@ static int create_erlang_module(void) { PyDict_SetItemString(sd_globals, "__builtins__", builtins); /* Import erlang module into globals so the code can reference it */ - PyObject *sys_modules = PySys_GetObject("modules"); + PyObject *sys_modules = PyImport_GetModuleDict(); /* the interpreter's table, a dict even when sys.modules is replaced */ if (sys_modules != NULL) { PyObject *erlang_mod = PyDict_GetItemString(sys_modules, "erlang"); if (erlang_mod != NULL) { From 9ddbb279b0c86281e5a0c34a273ab51dd65a7495 Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Sat, 26 Sep 2026 11:51:12 +0200 Subject: [PATCH 10/18] Run a function in a fresh module dictionary on an embedded context A template with start => reimport keeps worker or owngil contexts, and py_session:run/5 imports the function's module again in a new module dictionary on one of them, the way Temporal's Python SDK isolates a workflow run. The standard library, imports and passthrough modules are shared; the swap is per thread, so worker contexts run at once without seeing each other's modules. It isolates module state only, and the guide says what stays shared. --- CHANGELOG.md | 6 +- docs/code-map.md | 3 +- docs/decisions/0009-isolated-sessions.md | 15 +- docs/sessions.md | 55 +++++++- examples/bench_sessions.erl | 34 +++++ priv/_erlang_impl/_reimport.py | 168 +++++++++++++++++++++++ src/py_session.erl | 40 +++++- src/py_session_template.erl | 101 +++++++++++++- test/py_session_SUITE.erl | 143 ++++++++++++++++++- test/py_test_reimport.py | 69 ++++++++++ test/py_test_reimport_shared.py | 16 +++ 11 files changed, 633 insertions(+), 17 deletions(-) create mode 100644 priv/_erlang_impl/_reimport.py create mode 100644 test/py_test_reimport.py create mode 100644 test/py_test_reimport_shared.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 126d67a..096cb34 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -18,7 +18,11 @@ unchanged. `close/1` kills it; a session whose process dies answers with the reason until closed and is never restarted. `run/5` runs one call in a new session; `refresh/1` rebuilds the template after a deploy; `info/1` - reports zygotes, live sessions and forks. See `docs/sessions.md`. + reports zygotes, live sessions and forks. `start => reimport` runs each + call in a fresh module dictionary on a worker or owngil context instead, + the way Temporal's Python SDK isolates a workflow run: the function's + module is imported again per run, and the standard library, `imports` and + `passthrough` modules are shared. See `docs/sessions.md`. - `clear_env` and `hash_seed` options for isolated contexts: the child sees only the variables named in `env`, and every child built with the same seed hashes strings and orders sets the same way. diff --git a/docs/code-map.md b/docs/code-map.md index c8b1111..0677683 100644 --- a/docs/code-map.md +++ b/docs/code-map.md @@ -17,7 +17,7 @@ exercised by suites). Guides are in `docs/`, suites in `test/`. Start with | `py_context_embedded` | Process body for `worker` and `owngil` mode: the receive loop, callbacks (suspension and pipe), worker loops | live | architecture, state-machines | same | | `py_isolated` | `gen_statem` driving a child process over the socket; restart policy | live | isolated | `py_isolated_*_SUITE` | | `py_child` | Helpers shared by the processes that drive a Python child: executable lookup, socket listen/accept, frames, port env, rlimit flags | live | isolated | `py_isolated_*_SUITE` | -| `py_session` | Isolated sessions: a fresh child per session from a template (`template/1`, `new/1`, `close/1`, `run/4`, `refresh/1`) | live | sessions | `py_session_SUITE` | +| `py_session` | Isolated sessions: a fresh child per session from a template (`template/1`, `new/1`, `close/1`, `run/4`, `refresh/1`), or a fresh module dictionary per run (`start => reimport`) | live | sessions | `py_session_SUITE` | | `py_session_template`, `py_session_sup` | A template: zygotes that fork sessions (`priv/py_zygote.py`) or warm spawned sessions; exit reports to the session contexts | live | sessions | `py_session_SUITE` | | `py_context_router` | Pools and scheduler-affinity routing | live | pools, context-affinity | `py_context_router_SUITE`, `py_pool_SUITE` | | `py_context_sup`, `py_context_init` | Supervisor of contexts; starts the default pool at boot | live | pools | (through the above) | @@ -82,6 +82,7 @@ loop, channels and servers. | `_erlang_impl/_mode.py` | Detects how Python is running (embedded, free-threaded, child) | all | | `_erlang_impl/_etf.py` | Pure-Python ETF codec with the `py_convert.c` mapping | isolated child | | `_erlang_impl/_isolated.py` | Child runtime: socket frames, reader thread, re-entrant main loop, interrupt signal, asyncio loop, the `erlang` shim | isolated child | +| `_erlang_impl/_reimport.py` | Re-import runs for `py_session` templates with `start => reimport`: a fresh module dictionary per run, swapped per thread | embedded (worker, owngil) | | `_erlang_impl/_shm.py` | `SharedMemory` and `SharedBuffer` wrappers over mmap | all | | `py_isolated_child.py` | Child launcher: rlimits, parent-death signal, cgroup join, connect | isolated child | | `py_zygote.py` | Session template zygote: imports and preload once, forks one child per session, reports exits | session zygote | diff --git a/docs/decisions/0009-isolated-sessions.md b/docs/decisions/0009-isolated-sessions.md index 13fd9c3..653043f 100644 --- a/docs/decisions/0009-isolated-sessions.md +++ b/docs/decisions/0009-isolated-sessions.md @@ -31,11 +31,15 @@ request with the reason until it is closed. It runs in its own scratch directory, its stdio is detached from the zygote's port, and it sees only the template's environment. +`start => reimport` is the light variant for worker and owngil contexts: +no process per session, the function's module imported again in a fresh +module dictionary per run, swapped per thread as Temporal's workflow +sandbox does. It isolates module state only and is offered as that. + Not chosen: a subinterpreter per session (about 13 ms, shares the process environment, working directory, hash seed and C-extension state, and PyO3 -extensions refuse to load), Temporal-style re-import in one process (state -leaks through passthrough and C modules), CRIU (Linux only, needs -privileges and PID namespaces), and Wasm images (no native C extensions). +extensions refuse to load), CRIU (Linux only, needs privileges and PID +namespaces), and Wasm images (no native C extensions). ## Consequences @@ -50,3 +54,8 @@ privileges and PID namespaces), and Wasm images (no native C extensions). until a session exists. - A zygote that dies is rebuilt; its orphaned sessions are watched by pid (`py_isolated` probes `kill(pid, 0)`), since nobody reports their exit. +- A re-import template replaces `sys.modules` and `builtins.__import__` in + its interpreter with per-thread stand-ins. C code that reads the + interpreter's own module table (the C `pickle`) does not see a run's + modules; the NIF's own lookups use `PyImport_GetModuleDict()` for that + reason. diff --git a/docs/sessions.md b/docs/sessions.md index 2064813..8f9b558 100644 --- a/docs/sessions.md +++ b/docs/sessions.md @@ -54,6 +54,7 @@ fails. |---|---|---| | `fork` (default) | forked from the template's zygote: imports and preload are already done | the template can be forked | | `spawn` | a new interpreter runs the imports and preload | a module starts a thread when imported, or loads Objective-C on macOS | +| `reimport` | no new process: the function's module is imported again in a fresh module dictionary, in a worker or owngil context | you only need fresh module state per run, and each run is one call (see below) | A fork copies only the thread that calls it, so a template whose imports start a thread is refused: @@ -73,6 +74,54 @@ not wait for an interpreter: With `start => fork`, `zygotes => N` runs N zygotes that fork in turn when one is not enough. +## Re-import runs in a worker or owngil context + +When a fresh module state per run is enough, and you do not need a process +boundary, run in an embedded context instead. This is how Temporal's Python +SDK isolates a workflow run: + +```erlang +{ok, T} = py_session:template(#{ + start => reimport, + mode => worker, %% or owngil + contexts => 4, %% runs are spread over these + paths => ["/srv/app"], + imports => [orders_models], %% imported once, shared by every run + passthrough => [pydantic] %% shared too, imported when first used +}), +{ok, Result} = py_session:run(T, orders, handle, [Event, State]). +``` + +Each `run/5` imports `orders` again in a new module dictionary, so its +globals start fresh and nothing a run leaves in them reaches the next one. +The standard library, `erlang`, `imports` and `passthrough` modules are +shared with the context. A run is one call: `py_session:new/1` answers +`{error, {not_supported, reimport}}`. Calls back into the same context from +a callback work. + +| | `fork` / `spawn` | `reimport` | +|---|---|---| +| Module globals of the function's module and what it imports | fresh | fresh | +| Standard library, `imports`, `passthrough` | fresh (`fork`: as prepared) | shared, their state persists | +| C extension state, environment, working directory, threads, hash seed | fresh (hash seed per template) | shared with the context | +| A call stuck in C | killed | runs on; interrupts land at the next bytecode | +| A segfault in a C extension | kills the session | kills the node | +| Memory and CPU limits | yes | no | +| Cost per run | a fork, or a warm child | an import of the function's module | +| Cost per call inside the run | a local socket round trip | none | + +Two things to know about re-import runs: + +- `sys.modules` in that interpreter becomes a mapping that shows each thread + its run's modules, and `builtins.__import__` is replaced, from the first + re-import template on. Code that checks `type(sys.modules) is dict` sees + the difference. +- C code that looks modules up in the interpreter's own table does not see a + run's modules. The C `pickle` is one: pickling an instance of a class + defined in a re-imported module fails (`KeyError` or `PicklingError`, + depending on the Python version). Put such classes in + a module listed in `imports`, or use `start => fork`. + ## What a session sees | | Sessions of one template | @@ -139,12 +188,16 @@ with the size of the prepared process. | How the session starts | macOS 27, Python 3.14 | Linux (container), Python 3.11 | |---|---|---| +| `reimport` (one run, worker or owngil context) | ~1 ms | | | `fork` | ~4 ms | ~4 ms | | `spawn` with a `warm` pool that keeps up | ~2 ms | ~2 ms | | `spawn` | ~60 ms | ~45 ms | | a plain isolated context, for reference | ~60 ms | ~40 ms | -One zygote forks one session at a time. On Linux it served about 1,000 +Re-import runs on worker contexts share one GIL; on owngil contexts they +run in parallel (about 3,500 runs a second with 8 callers on four +contexts, against about 900 on worker contexts). One zygote forks one +session at a time. On Linux it served about 1,000 sessions a second with 8 or more callers; add `zygotes` when sessions are opened faster than one zygote forks them. diff --git a/examples/bench_sessions.erl b/examples/bench_sessions.erl index f10219a..5fb15bf 100644 --- a/examples/bench_sessions.erl +++ b/examples/bench_sessions.erl @@ -9,6 +9,7 @@ %%% template, spawn template with a warm pool, plain isolated context %%% 2. sessions per second with 1, 8 and 64 callers %%% 3. a call and a re-entrant call (Python -> Erlang -> same session) +%%% 4. a re-import run (start => reimport) on worker and owngil contexts %%% %%% examples/bench_sessions_sdks.py measures what Temporal's workflow %%% sandbox and Restate's per-invocation state cost on the same module, to @@ -68,8 +69,41 @@ main(_) -> row("re-entrant call", [us(fun() -> {ok, 1} = py_context:call(S, bench_sessions_wf, reenter, [1]) end) || _ <- lists:seq(1, 1000)]), py_session:close(S), + io:format("~n== re-import runs (no process per run) ==~n"), + %% the same workflow as a module file: a re-import run imports it anew + Dir = filename:join(py_child:sock_dir(), "bench_reimport"), + ok = filelib:ensure_dir(filename:join(Dir, "x")), + ok = file:write_file(filename:join(Dir, "bench_reimport_wf.py"), reimport_module()), + [begin + {ok, R} = py_session:template(#{start => reimport, mode => M, contexts => 4, + paths => [Dir]}), + row(atom_to_list(M) ++ ", one run", + [us(fun() -> {ok, _} = py_session:run(R, bench_reimport_wf, handle, [[], #{}]) end) + || _ <- lists:seq(1, 500)]), + L = reimport_throughput(R, 8, 3), + io:format(" ~-18s ~7.1f runs/s with 8 callers~n", [M, length(L) / 3]) + end || M <- [worker] ++ [owngil || py_nif:owngil_supported()]], ok. +reimport_module() -> + <<"from dataclasses import dataclass\nimport json, decimal, datetime\n\n" + "@dataclass\nclass Step:\n n: int\n\n" + "def handle(event, state):\n" + " n = state.get('n', 0) + 1\n" + " return {'commands': [['activity', 'charge', {'n': n}]], 'state': {'n': Step(n).n}}\n">>. + +reimport_throughput(T, Callers, Secs) -> + Self = self(), + Deadline = erlang:monotonic_time(millisecond) + Secs * 1000, + Loop = fun L(Acc) -> + case erlang:monotonic_time(millisecond) < Deadline of + true -> L([us(fun() -> {ok, _} = py_session:run(T, bench_reimport_wf, handle, [[], #{}]) end) | Acc]); + false -> Acc + end + end, + Pids = [spawn_link(fun() -> Self ! {done, self(), Loop([])} end) || _ <- lists:seq(1, Callers)], + lists:append([receive {done, P, L} -> L end || P <- Pids]). + template(Opts) -> {ok, T} = py_session:template(Opts#{preload => ?PY}), T. diff --git a/priv/_erlang_impl/_reimport.py b/priv/_erlang_impl/_reimport.py new file mode 100644 index 0000000..6195b30 --- /dev/null +++ b/priv/_erlang_impl/_reimport.py @@ -0,0 +1,168 @@ +# Copyright 2026 Benoit Chesneau +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +""" +Re-import runs for sessions started with `start => reimport`. + +Each run gets its own module dictionary: the modules the template passes +through (the standard library, `erlang`, and the names it lists) are shared +with the interpreter, every other module is imported again inside the run, +so module globals start fresh and nothing a run changes in them is seen by +the next one. This is the isolation of Temporal's Python workflow sandbox, +without its restrictions on time, randomness and I/O. + +The swap is per thread. `sys.modules` and `builtins.__import__` are replaced +once, in each interpreter, by stand-ins that use the running thread's module +dictionary when it has one and the interpreter's otherwise. A worker context +is one thread of the main interpreter, so several worker contexts run their +sandboxes at once without seeing each other's. + +Passthrough modules are imported by the interpreter itself, not inside the +run: C code looks modules up in the interpreter's own table and must find +them there. +""" + +import builtins +import importlib +import sys +import threading +from collections.abc import MutableMapping + +_tls = threading.local() +_install_lock = threading.Lock() +_real_import = builtins.__import__ +_real_modules = sys.modules + +_ALWAYS_SHARED = frozenset(sys.stdlib_module_names) | {'erlang', '_erlang_impl', 'py_event_loop'} + + +class _ThreadModules(MutableMapping): + """sys.modules as seen by each thread: its run's dictionary while a run + is active, the interpreter's otherwise. A mapping rather than a dict + subclass, so C code that reads sys.modules goes through these methods + instead of an empty dict storage.""" + + def _d(self): + mods = getattr(_tls, 'modules', None) + return _real_modules if mods is None else mods + + def __getitem__(self, key): + return self._d()[key] + + def __setitem__(self, key, value): + self._d()[key] = value + + def __delitem__(self, key): + del self._d()[key] + + def __contains__(self, key): + return key in self._d() + + def __iter__(self): + return iter(list(self._d())) + + def __len__(self): + return len(self._d()) + + def get(self, key, default=None): + return self._d().get(key, default) + + def copy(self): + return dict(self._d()) + + def __repr__(self): + return repr(self._d()) + + +def _install(): + global _real_modules + with _install_lock: + if not isinstance(sys.modules, _ThreadModules): + _real_modules = sys.modules + sys.modules = _ThreadModules() + builtins.__import__ = _import + + +def _import(name, globals=None, locals=None, fromlist=(), level=0): + mods = getattr(_tls, 'modules', None) + if mods is None: + return _real_import(name, globals, locals, fromlist, level) + sandbox = _tls.sandbox + if level == 0 and sandbox.shared(name) and name not in mods: + sandbox.pass_through(name, fromlist, mods) + return importlib.__import__(name, globals, locals, fromlist, level) + + +class Sandbox: + """Runs functions in fresh module dictionaries. `passthrough` names + (top-level packages) are shared with the interpreter.""" + + def __init__(self, passthrough=()): + _install() + self.passthrough = frozenset(passthrough) + + def shared(self, name): + root = name.split('.', 1)[0] + return root in _ALWAYS_SHARED or root in self.passthrough + + def pass_through(self, name, fromlist, mods): + """Import `name` in the interpreter, then share its module objects + (with its parents and loaded submodules) with the run.""" + _tls.modules = None + try: + _real_import(name, None, None, fromlist or (), 0) + for sub in fromlist or (): + full = name + '.' + sub + if full not in _real_modules: + try: + _real_import(full) + except ImportError: + pass # an attribute, not a submodule + finally: + _tls.modules = mods + parts = name.split('.') + for i in range(1, len(parts) + 1): + parent = '.'.join(parts[:i]) + if parent in _real_modules: + mods[parent] = _real_modules[parent] + prefix = name + '.' + for key, mod in list(_real_modules.items()): + if key.startswith(prefix) and key not in mods: + mods[key] = mod + + def run(self, module, func, args=(), kwargs=None): + prev = (getattr(_tls, 'modules', None), getattr(_tls, 'sandbox', None)) + _tls.sandbox = self + _tls.modules = {k: m for k, m in list(_real_modules.items()) if self.shared(k)} + try: + fn = getattr(importlib.import_module(module), func) + return fn(*args, **(kwargs or {})) + finally: + _tls.modules, _tls.sandbox = prev + + +_sandboxes = {} + + +def run(passthrough, module, func, args, kwargs): + """Entry point called by py_session:run/5 on a reimport template.""" + key = tuple(sorted(_as_text(p) for p in passthrough)) + sandbox = _sandboxes.get(key) + if sandbox is None: + sandbox = _sandboxes[key] = Sandbox(key) + return sandbox.run(_as_text(module), _as_text(func), list(args), dict(kwargs or {})) + + +def _as_text(v): + return v.decode('utf-8') if isinstance(v, (bytes, bytearray)) else str(v) diff --git a/src/py_session.erl b/src/py_session.erl index 2578ef7..7b22a5f 100644 --- a/src/py_session.erl +++ b/src/py_session.erl @@ -37,7 +37,12 @@ %%%
  • `start' - `fork' (default): sessions are forked from a zygote that %%% already ran the imports and preload. `spawn': each session starts %%% a new interpreter; use it when the template cannot be forked -%%% (threads started at import, Objective-C on macOS).
  • +%%% (threads started at import, Objective-C on macOS). `reimport': +%%% no process per session; run/5 imports the function's module +%%% again in a fresh module dictionary on one of the template's +%%% `contexts' (`mode => worker | owngil'), as Temporal's workflow +%%% sandbox does. Modules in `imports' and `passthrough', and the +%%% standard library, are shared by every run. %%%
  • `python', `paths', `imports', `preload' - the prepared state.
  • %%%
  • `env' - the environment of every session. Nothing is inherited %%% from the VM unless `clear_env => false'.
  • @@ -100,6 +105,9 @@ new(T, Opts) -> end; {start, CtxOpts} -> py_context:new(maps:merge(CtxOpts, maps:with([start_timeout], Opts))); + {reimport, _, _} -> + %% a re-import run is not a process: use run/5 + {error, {not_supported, reimport}}; {error, _} = Err -> Err end. @@ -120,6 +128,36 @@ run(T, Module, Func, Args) -> -spec run(template(), atom() | binary(), atom() | binary(), list(), map()) -> {ok, term()} | {error, term()}. run(T, Module, Func, Args, Opts) -> + case py_session_template:checkout(T, maps:get(timeout, Opts, 15000)) of + {reimport, Ctx, Passthrough} -> + py_context:call(Ctx, '_erlang_impl._reimport', run, + [Passthrough, py_child:to_bin(Module), py_child:to_bin(Func), + Args, maps:get(kwargs, Opts, #{})], + #{}, maps:get(timeout, Opts, infinity)); + {ok, Ctx} -> + %% taken from the warm pool, handed over as new/2 does + run_in(T, Ctx, Module, Func, Args, Opts); + _ -> + run_new(T, Module, Func, Args, Opts) + end. + +run_in(T, Ctx, Module, Func, Args, Opts) -> + MRef = erlang:monitor(process, Ctx), + Ctx ! {set_parent, self(), MRef, self()}, + receive + {MRef, ok} -> + erlang:demonitor(MRef, [flush]), + try + py_context:call(Ctx, Module, Func, Args, maps:get(kwargs, Opts, #{}), + maps:get(timeout, Opts, infinity)) + after + close(Ctx) + end; + {'DOWN', MRef, process, Ctx, _} -> + run(T, Module, Func, Args, Opts) + end. + +run_new(T, Module, Func, Args, Opts) -> case new(T, Opts) of {ok, S} -> try diff --git a/src/py_session_template.erl b/src/py_session_template.erl index 0c54cdc..c30d0c5 100644 --- a/src/py_session_template.erl +++ b/src/py_session_template.erl @@ -23,6 +23,11 @@ %%% `start => spawn' starts each session as a new isolated child and can keep %%% `warm' of them started ahead of time; each is handed out once. %%% +%%% `start => reimport' keeps `contexts' worker or owngil contexts; each +%%% py_session:run/5 runs in a fresh module dictionary on one of them +%%% (priv/_erlang_impl/_reimport.py), in the way of Temporal's workflow +%%% sandbox. There is no session process: new/1 is not available. +%%% %%% A crashed zygote is rebuilt; the sessions it forked are separate %%% processes and keep running. %%% @@ -62,7 +67,10 @@ -record(st, { opts :: map(), - start :: fork | spawn, + start :: fork | spawn | reimport, + %% start => reimport: the contexts runs go to, and what they share + contexts = [] :: [pid()], + passthrough = [] :: [binary()], zygotes = [] :: [#zygote{}], next = 0 :: non_neg_integer(), next_id = 1 :: pos_integer(), @@ -98,7 +106,8 @@ fork(T, SockPath, ForkOpts, Timeout) -> %% @doc A context started ahead of time (`{ok, Ctx}', linked to the %% template until handed over), or the options to start one. --spec checkout(pid(), timeout()) -> {ok, pid()} | {start, map()} | {error, term()}. +-spec checkout(pid(), timeout()) -> + {ok, pid()} | {start, map()} | {reimport, pid(), [binary()]} | {error, term()}. checkout(T, Timeout) -> try gen_server:call(T, checkout, Timeout) catch @@ -146,6 +155,9 @@ handle_call({fork, Path, ForkOpts}, {Pid, _} = From, St) -> {error, Reason} -> {reply, {error, {zygote_unreachable, Reason}}, St} end; +handle_call(checkout, _From, #st{start = reimport, contexts = Cs, next = N, + passthrough = PT} = St) -> + {reply, {reimport, lists:nth(N rem length(Cs) + 1, Cs), PT}, St#st{next = N + 1}}; handle_call(checkout, _From, #st{start = fork} = St) -> {reply, {start, context_opts(St)}, St}; handle_call(checkout, _From, #st{warm = [Ctx | Rest]} = St) -> @@ -164,6 +176,14 @@ handle_call(refresh, _From, #st{start = fork, zygotes = Old} = St) -> {error, Reason} -> {reply, {error, Reason}, St} end; +handle_call(refresh, _From, #st{start = reimport, contexts = Old} = St) -> + case build(St#st{contexts = []}) of + {ok, St1} -> + [begin unlink(C), py_context:stop(C) end || C <- Old], + {reply, ok, St1}; + {error, Reason} -> + {reply, {error, Reason}, St} + end; handle_call(refresh, _From, #st{start = spawn, warm = Warm} = St) -> [begin unlink(C), py_context:stop(C) end || C <- Warm], self() ! fill_warm, @@ -211,15 +231,30 @@ handle_info(fill_warm, #st{start = spawn, warm = Warm, opts = Opts} = St) -> end; handle_info(fill_warm, St) -> {noreply, St}; +handle_info({'EXIT', Pid, Reason}, #st{start = reimport, contexts = Cs} = St) -> + case lists:member(Pid, Cs) of + true -> + %% A context of the template stopped: put a new one in its place + logger:warning("py_session template ~p: context ~p exited: ~p", + [self(), Pid, Reason]), + case start_context(St#st.opts) of + {ok, C} -> + {noreply, St#st{contexts = [C | lists:delete(Pid, Cs)]}}; + {error, Why} -> + {stop, {context_restart_failed, Why}, St} + end; + false -> + {noreply, St} + end; handle_info({'EXIT', Pid, _Reason}, #st{warm = Warm} = St) -> %% A warm session died before anyone took it {noreply, St#st{warm = lists:delete(Pid, Warm)}}; handle_info(_Msg, St) -> {noreply, St}. -terminate(_Reason, #st{zygotes = Zs, warm = Warm}) -> +terminate(_Reason, #st{zygotes = Zs, warm = Warm, contexts = Cs}) -> [retire(Z) || Z <- Zs], - [py_context:stop(C) || C <- Warm], + [py_context:stop(C) || C <- Warm ++ Cs], ok. %% ============================================================================ @@ -229,9 +264,18 @@ terminate(_Reason, #st{zygotes = Zs, warm = Warm}) -> check_opts(Opts) -> Checks = [ fun() -> case maps:get(start, Opts, fork) of - S when S =:= fork; S =:= spawn -> ok; + S when S =:= fork; S =:= spawn; S =:= reimport -> ok; S -> {error, {badarg, {start, S}}} end end, + fun() -> case {maps:get(start, Opts, fork), maps:get(mode, Opts, worker)} of + {reimport, M} when M =:= worker; M =:= owngil -> ok; + {reimport, M} -> {error, {badarg, {mode, M}}}; + _ -> ok + end end, + fun() -> case maps:get(contexts, Opts, 1) of + N when is_integer(N), N >= 1 -> ok; + N -> {error, {badarg, {contexts, N}}} + end end, fun() -> case maps:get(zygotes, Opts, 1) of N when is_integer(N), N >= 1 -> ok; N -> {error, {badarg, {zygotes, N}}} @@ -260,6 +304,19 @@ env_opts(Opts) -> build(#st{start = spawn} = St) -> self() ! fill_warm, {ok, St}; +build(#st{start = reimport, opts = Opts} = St) -> + T0 = erlang:monotonic_time(millisecond), + Started = [start_context(Opts) || _ <- lists:seq(1, maps:get(contexts, Opts, 1))], + case [E || {error, _} = E <- Started] of + [] -> + PT = lists:usort([py_child:to_bin(M) || M <- maps:get(imports, Opts, []) + ++ maps:get(passthrough, Opts, [])]), + {ok, St#st{contexts = [C || {ok, C} <- Started], passthrough = PT, + build_ms = erlang:monotonic_time(millisecond) - T0}}; + [Err | _] -> + [py_context:stop(C) || {ok, C} <- Started], + Err + end; build(#st{opts = Opts} = St) -> T0 = erlang:monotonic_time(millisecond), case build_zygotes(maps:get(zygotes, Opts, 1), St#st.opts, []) of @@ -317,6 +374,37 @@ start_zygote(Opts) -> {error, {template_failed, Reason}} end. +%% A context for re-import runs: the paths, the imports (shared by every +%% run) and the preload are applied once. +start_context(Opts) -> + CtxOpts = maps:merge(#{mode => maps:get(mode, Opts, worker)}, + maps:with([preload], Opts)), + case py_context:new(CtxOpts) of + {ok, C} -> + Paths = [py_child:to_list(P) || P <- maps:get(paths, Opts, [])], + Setup = iolist_to_binary(io_lib:format( + "import sys, importlib +" + "for _p in reversed(~p): +" + " if _p not in sys.path: sys.path.insert(0, _p) +" + "for _m in ~p: importlib.import_module(_m) +" + "import _erlang_impl._reimport +", + [Paths, [py_child:to_list(M) || M <- maps:get(imports, Opts, [])]])), + case py_context:exec(C, Setup) of + ok -> + {ok, C}; + {error, Reason} -> + py_context:stop(C), + {error, {template_failed, {init_failed, Reason}}} + end; + {error, Reason} -> + {error, {template_failed, Reason}} + end. + %% Wait for `ready', run the imports and preload, then arm the socket. prepare(Z, Opts, Timeout) -> case recv_sync(Z, Timeout) of @@ -494,8 +582,9 @@ fork_opts(ForkOpts) -> end, #{}, ForkOpts). info_map(#st{start = Start, zygotes = Zs, owners = Owners, forks = Forks, - build_ms = BuildMs, warm = Warm, opts = Opts}) -> + build_ms = BuildMs, warm = Warm, opts = Opts, contexts = Cs}) -> #{start => Start, + contexts => Cs, zygotes => [Info#{os_pid => P} || #zygote{os_pid = P, info = Info} <- Zs], sessions => map_size(Owners), forks => Forks, diff --git a/test/py_session_SUITE.erl b/test/py_session_SUITE.erl index 07f0262..f09d4c7 100644 --- a/test/py_session_SUITE.erl +++ b/test/py_session_SUITE.erl @@ -38,13 +38,23 @@ test_erlang_call_during_preload/1, test_info_counts_forks/1, test_warm_pool/1, - test_thread_at_import_spawned/1 + test_thread_at_import_spawned/1, + test_reimport_runs_are_isolated/1, + test_reimport_shares_stdlib_and_imports/1, + test_reimport_concurrent_runs/1, + test_reimport_reentrance/1, + test_reimport_typing_and_pickle/1, + test_reimport_errors/1, + test_reimport_no_session_process/1, + test_reimport_refresh/1, + test_reimport_bad_options/1 ]). -define(MOD, py_test_session). all() -> - [{group, fork}, {group, spawn}, {group, fork_only}, {group, spawn_only}]. + [{group, fork}, {group, spawn}, {group, fork_only}, {group, spawn_only}, + {group, reimport_worker}, {group, reimport_owngil}]. groups() -> Common = [ @@ -68,6 +78,17 @@ groups() -> test_refresh_picks_new_code, test_bad_template_options ], + Reimport = [ + test_reimport_runs_are_isolated, + test_reimport_shares_stdlib_and_imports, + test_reimport_concurrent_runs, + test_reimport_reentrance, + test_reimport_typing_and_pickle, + test_reimport_errors, + test_reimport_no_session_process, + test_reimport_refresh, + test_reimport_bad_options + ], [{fork, [], Common}, {spawn, [], Common}, {fork_only, [], [test_zygote_crash_rebuilds, @@ -75,7 +96,9 @@ groups() -> test_erlang_call_during_preload, test_info_counts_forks]}, {spawn_only, [], [test_warm_pool, - test_thread_at_import_spawned]}]. + test_thread_at_import_spawned]}, + {reimport_worker, [], Reimport}, + {reimport_owngil, [], Reimport}]. init_per_suite(Config) -> {ok, _} = application:ensure_all_started(erlang_python), @@ -96,6 +119,13 @@ end_per_suite(_Config) -> init_per_group(G, Config) when G =:= fork; G =:= fork_only -> [{start, fork} | Config]; +init_per_group(reimport_worker, Config) -> + [{start, reimport}, {mode, worker} | Config]; +init_per_group(reimport_owngil, Config) -> + case py_nif:owngil_supported() of + true -> [{start, reimport}, {mode, owngil} | Config]; + false -> {skip, "OWN_GIL requires Python 3.14+"} + end; init_per_group(_G, Config) -> [{start, spawn} | Config]. @@ -291,8 +321,10 @@ test_session_crash(Config) -> {ok, 1} = py_context:eval(B, <<"1">>), {ok, C} = py_session:new(T), {ok, 1} = py_context:eval(C, <<"1">>), + MA = erlang:monitor(process, A), [py_session:close(X) || X <- [A, B, C]], - false = is_process_alive(A), + %% close/1 replies first, then the context removes its directory and exits + receive {'DOWN', MA, process, A, _} -> ok after 5000 -> ct:fail(session_not_stopped) end, ok. test_kill_session(Config) -> @@ -499,6 +531,109 @@ test_thread_at_import_spawned(Config) -> {ok, 2} = py_context:call(S, ?MOD, thread_count, []), py_session:close(S). +%%% ============================================================================ +%%% reimport +%%% ============================================================================ + +-define(RMOD, py_test_reimport). + +%% Each run imports the module again: its globals start fresh every time, +%% and the interpreter's own copy is never touched. +test_reimport_runs_are_isolated(Config) -> + T = rtemplate(Config), + [{ok, 1} = py_session:run(T, ?RMOD, bump, []) || _ <- lists:seq(1, 20)], + {ok, <<"py_test_reimport">>} = py_session:run(T, ?RMOD, whoami, []), + %% the context itself never imported it + #{contexts := [C | _]} = py_session:info(T), + {ok, false} = py_context:eval(C, <<"'py_test_reimport' in __import__('sys').modules">>), + ok. + +%% What is shared on purpose: the standard library and `imports'. +test_reimport_shares_stdlib_and_imports(Config) -> + T = rtemplate(Config, #{contexts => 1, imports => [py_test_reimport_shared]}), + {ok, 1} = py_session:run(T, ?RMOD, bump_shared, []), + {ok, 2} = py_session:run(T, ?RMOD, bump_shared, []), + {ok, <<"m">>} = py_session:run(T, ?RMOD, mark_stdlib, [<<"m">>]), + {ok, <<"m">>} = py_session:run(T, ?RMOD, read_stdlib_mark, []), + {ok, 1} = py_session:run(T, ?RMOD, bump, []), + ok. + +%% Contexts run sandboxes at the same time without seeing each other's. +test_reimport_concurrent_runs(Config) -> + T = rtemplate(Config, #{contexts => 4}), + Self = self(), + Pids = [spawn_link(fun() -> + Self ! {done, [py_session:run(T, ?RMOD, bump, []) || _ <- lists:seq(1, 50)]} + end) || _ <- lists:seq(1, 8)], + [receive {done, R} -> R = lists:duplicate(50, {ok, 1}) after 60000 -> ct:fail(timeout) end + || _ <- Pids], + ok. + +%% A run calls Erlang, whose callback runs another re-import run on the +%% same template (one context: the nested run lands on the waiting one). +test_reimport_reentrance(Config) -> + T = rtemplate(Config, #{contexts => 1}), + py:register_function(reimport_nested, fun([N]) -> + {ok, R} = py_session:run(T, ?RMOD, bump, [], #{timeout => 10000}), + N + R + end), + try + [{ok, 11} = py_session:run(T, ?RMOD, call_back, [10], #{timeout => 10000}) + || _ <- lists:seq(1, 10)], + {ok, 1} = py_session:run(T, ?RMOD, bump, []) + after + py:unregister_function(reimport_nested) + end. + +%% Python code that reads sys.modules (typing) finds the run's module. C +%% code that reads the interpreter's own table (the C pickle) does not: +%% a class to pickle belongs in a shared module. +test_reimport_typing_and_pickle(Config) -> + T = rtemplate(Config, #{imports => [py_test_reimport_shared]}), + {ok, [<<"x">>, <<"y">>]} = py_session:run(T, ?RMOD, typed, []), + {ok, 3} = py_session:run(T, ?RMOD, pickle_shared, []), + %% KeyError on 3.14, PicklingError on 3.11: either way it fails + {ok, Err} = py_session:run(T, ?RMOD, pickle_reimported, []), + true = lists:member(Err, [<<"KeyError">>, <<"PicklingError">>]), + ok. + +test_reimport_errors(Config) -> + T = rtemplate(Config), + {error, {'ValueError', _}} = py_session:run(T, ?RMOD, fail, [<<"boom">>]), + {error, _} = py_session:run(T, no_such_module_xyz, f, []), + {ok, 1} = py_session:run(T, ?RMOD, bump, []), + ok. + +test_reimport_no_session_process(Config) -> + T = rtemplate(Config), + {error, {not_supported, reimport}} = py_session:new(T), + ok. + +test_reimport_refresh(Config) -> + T = rtemplate(Config, #{contexts => 2}), + #{contexts := Old} = py_session:info(T), + ok = py_session:refresh(T), + #{contexts := New} = py_session:info(T), + [] = [C || C <- New, lists:member(C, Old)], + {ok, 1} = py_session:run(T, ?RMOD, bump, []), + ok. + +test_reimport_bad_options(Config) -> + {error, {badarg, {mode, isolated}}} = + py_session:template(#{start => reimport, mode => isolated}), + {error, {badarg, {contexts, 0}}} = + py_session:template(#{start => reimport, mode => ?config(mode, Config), contexts => 0}), + ok. + +rtemplate(Config) -> + rtemplate(Config, #{}). + +rtemplate(Config, Extra) -> + {ok, T} = py_session:template(maps:merge(#{start => reimport, mode => ?config(mode, Config), + contexts => 2, + paths => [?config(test_dir, Config)]}, Extra)), + T. + %%% ============================================================================ %%% Helpers %%% ============================================================================ diff --git a/test/py_test_reimport.py b/test/py_test_reimport.py new file mode 100644 index 0000000..66a506c --- /dev/null +++ b/test/py_test_reimport.py @@ -0,0 +1,69 @@ +"""Imported again in every re-import run (py_session_SUITE reimport groups).""" + +import json +import pickle +import typing +from dataclasses import dataclass + +import erlang +import py_test_reimport_shared as shared + +COUNT = {'n': 0} + + +@dataclass +class Point: + x: int + y: int + + +def bump(): + COUNT['n'] += 1 + return COUNT['n'] + + +def bump_shared(): + return shared.bump() + + +def mark_stdlib(value): + """The standard library is shared: this is seen by later runs.""" + json._reimport_mark = value + return value + + +def read_stdlib_mark(): + return getattr(json, '_reimport_mark', None) + + +def call_back(n): + return erlang.call('reimport_nested', n) + + +def typed(): + """typing reads sys.modules from Python: it finds the run's module.""" + return sorted(typing.get_type_hints(Point)) + + +def pickle_shared(): + """A class from a shared module pickles.""" + p = pickle.loads(pickle.dumps(shared.Pair(1, 2))) + return p.a + p.b + + +def pickle_reimported(): + """The C pickle looks the class's module up in the interpreter's own + table, where a re-imported module is not.""" + try: + pickle.dumps(Point(1, 2)) + except Exception as exc: + return type(exc).__name__ + return 'pickled' + + +def fail(msg): + raise ValueError(msg) + + +def whoami(): + return __name__ diff --git a/test/py_test_reimport_shared.py b/test/py_test_reimport_shared.py new file mode 100644 index 0000000..fc2f648 --- /dev/null +++ b/test/py_test_reimport_shared.py @@ -0,0 +1,16 @@ +"""Listed in a reimport template's `imports`: shared by every run.""" + +from dataclasses import dataclass + +COUNT = {'n': 0} + + +@dataclass +class Pair: + a: int + b: int + + +def bump(): + COUNT['n'] += 1 + return COUNT['n'] From b27cdc41b44a3988d552cc2b2f4ba1346fe1a8fd Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Sat, 26 Sep 2026 13:37:11 +0200 Subject: [PATCH 11/18] Wait for a forked session's exit report before guessing The zygote reaps a child and its template reports how it died; on a busy machine the pid disappears before that report arrives, and the session concluded it was killed. It now waits for the report while the template is alive, and falls back to the pid only when it is gone. --- src/py_isolated.erl | 18 ++++++++++++++++-- test/py_session_SUITE.erl | 15 ++++++++++++++- 2 files changed, 30 insertions(+), 3 deletions(-) diff --git a/src/py_isolated.erl b/src/py_isolated.erl index 3797eda..b2b0f93 100644 --- a/src/py_isolated.erl +++ b/src/py_isolated.erl @@ -94,6 +94,7 @@ -define(SHUTDOWN_GRACE_MS, 1000). -define(EXIT_STATUS_WAIT_MS, 5000). -define(EXIT_PROBE_MS, 100). +-define(EXIT_REPORT_GRACE_MS, 2000). -record(child, { %% undefined for a child forked by a session template @@ -226,10 +227,23 @@ handle_event(info, {py_session_exited, OsPid, Code}, State, handle_event(info, {py_session_exited, _, _}, _State, _Data) -> keep_state_and_data; handle_event(state_timeout, {probe_exit, Waited}, {restarting, _} = State, - #data{child = #child{os_pid = OsPid}} = Data) -> + #data{child = #child{os_pid = OsPid}, opts = Opts} = Data) -> case py_child:os_pid_alive(OsPid) of false -> - child_exited({signal, 9}, State, Data); + %% Reaped. Its template reports how it died, a little later + %% than the pid disappears when the machine is busy: wait for + %% that report unless nobody is left to send it. + ReportExpected = case maps:get(origin, Opts, spawn) of + {fork, T} -> is_process_alive(T) andalso Waited < ?EXIT_REPORT_GRACE_MS; + _ -> false + end, + case ReportExpected of + true -> + {keep_state_and_data, + [{state_timeout, ?EXIT_PROBE_MS, {probe_exit, Waited + ?EXIT_PROBE_MS}}]}; + false -> + child_exited({signal, 9}, State, Data) + end; true when Waited >= ?EXIT_STATUS_WAIT_MS -> handle_event(state_timeout, exit_status, State, Data); true -> diff --git a/test/py_session_SUITE.erl b/test/py_session_SUITE.erl index f09d4c7..458f37f 100644 --- a/test/py_session_SUITE.erl +++ b/test/py_session_SUITE.erl @@ -37,6 +37,7 @@ test_thread_at_import_refused/1, test_erlang_call_during_preload/1, test_info_counts_forks/1, + test_late_exit_report/1, test_warm_pool/1, test_thread_at_import_spawned/1, test_reimport_runs_are_isolated/1, @@ -94,7 +95,8 @@ groups() -> {fork_only, [], [test_zygote_crash_rebuilds, test_thread_at_import_refused, test_erlang_call_during_preload, - test_info_counts_forks]}, + test_info_counts_forks, + test_late_exit_report]}, {spawn_only, [], [test_warm_pool, test_thread_at_import_spawned]}, {reimport_worker, [], Reimport}, @@ -499,6 +501,17 @@ test_erlang_call_during_preload(Config) -> {match, _} = re:run(Msg, <<"not connected">>), py_session:close(S). +%% The template reports how a child died after the pid is gone; on a busy +%% machine that can take a while. The session must wait for the report +%% rather than guess "killed" from the pid. +test_late_exit_report(Config) -> + T = template(Config), + {ok, S} = py_session:new(T), + ok = sys:suspend(T), + spawn(fun() -> timer:sleep(700), sys:resume(T) end), + {error, {child_exited, {signal, 6}}} = py_context:call(S, ?MOD, abort, []), + py_session:close(S). + test_info_counts_forks(Config) -> T = template(Config, #{zygotes => 2}), #{start := fork, zygotes := [_, _], forks := 0} = py_session:info(T), From 5810f478c641f892ce5e84525d8c40f2133913a6 Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Sat, 26 Sep 2026 21:33:27 +0200 Subject: [PATCH 12/18] Give the measured cost of sessions in the guide Figures from examples/bench_sessions.erl on an unloaded machine, with throughput per start mode and the isolation step of Temporal's and Restate's SDKs on the same workflow. --- docs/sessions.md | 51 +++++++++++++++++++++++++++--------------------- 1 file changed, 29 insertions(+), 22 deletions(-) diff --git a/docs/sessions.md b/docs/sessions.md index 8f9b558..a67ee3f 100644 --- a/docs/sessions.md +++ b/docs/sessions.md @@ -182,29 +182,36 @@ started with. ## What it costs -Measured with `examples/bench_sessions.erl`, new + first call + close, -p50. Run it on your machine for your own figures: the cost of a fork grows -with the size of the prepared process. +Measured with `examples/bench_sessions.erl` on Apple silicon (macOS 27, +Python 3.14, 14 cores), p50 of new + first call + close. Run it on your +machine for your own figures: the cost of a fork grows with the size of +the prepared process. -| How the session starts | macOS 27, Python 3.14 | Linux (container), Python 3.11 | -|---|---|---| -| `reimport` (one run, worker or owngil context) | ~1 ms | | -| `fork` | ~4 ms | ~4 ms | -| `spawn` with a `warm` pool that keeps up | ~2 ms | ~2 ms | -| `spawn` | ~60 ms | ~45 ms | -| a plain isolated context, for reference | ~60 ms | ~40 ms | - -Re-import runs on worker contexts share one GIL; on owngil contexts they -run in parallel (about 3,500 runs a second with 8 callers on four -contexts, against about 900 on worker contexts). One zygote forks one -session at a time. On Linux it served about 1,000 -sessions a second with 8 or more callers; add `zygotes` when sessions are -opened faster than one zygote forks them. - -A call inside a session costs what it costs in any isolated context, since -it crosses the same socket. `examples/bench_sessions_sdks.py` measures the -isolation step of Temporal's workflow sandbox and Restate's SDK on the same -workflow module. +| How the session starts | Time to a used and closed session | +|---|---| +| `reimport` (one run, worker or owngil context) | 0.25 ms | +| `spawn` with a `warm` pool that keeps up | 0.6 ms | +| `fork` | 3.5 ms | +| `spawn` | 65 ms | +| a plain isolated context, for reference | 60 ms | + +| Throughput, 8 callers | | +|---|---| +| `fork`, one zygote | about 650 sessions a second (Linux: about 1,000) | +| `reimport` on four owngil contexts | about 13,500 runs a second | +| `reimport` on four worker contexts (one GIL) | about 3,500 runs a second | + +One zygote forks one session at a time; add `zygotes` when sessions are +opened faster than one zygote forks them. A call inside a session costs +what it costs in any isolated context (30 us p50, 70 us for a call that +calls back into the session), since it crosses the same socket. A call in +a re-import run stays in the process. + +`examples/bench_sessions_sdks.py` measures the isolation step of Temporal's +workflow sandbox (0.5 ms for a standard-library workflow, 4.6 ms when it +defines a pydantic model, since the module is imported again each run) and +of Restate's SDK (under a microsecond: it does not isolate invocations) on +the same workflow module. ## Limits From 17b899a7dfb5922c08bebf60ebc6243f16eab7d6 Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Wed, 30 Sep 2026 09:41:16 +0200 Subject: [PATCH 13/18] Send a message to any context that is not a NIF context interrupt, kill, pass_fd, child_info and submit asked a context process directly only when it registered as isolated. Any atom marker now means the same, so other kinds of context process can answer them. --- src/py_context.erl | 17 ++++++++++------- 1 file changed, 10 insertions(+), 7 deletions(-) diff --git a/src/py_context.erl b/src/py_context.erl index 757df51..6441dc9 100644 --- a/src/py_context.erl +++ b/src/py_context.erl @@ -424,7 +424,7 @@ get_nif_ref(Ctx) when is_pid(Ctx) -> -spec interrupt(context()) -> ok | not_running. interrupt(Ctx) when is_pid(Ctx) -> case lookup_nif_ref(Ctx) of - {ok, isolated} -> + {ok, Marker} when is_atom(Marker) -> %% The context process is never blocked in a NIF: ask it. It %% signals the child and arms the SIGKILL backstop. MRef = erlang:monitor(process, Ctx), @@ -457,7 +457,7 @@ interrupt(Ctx) when is_pid(Ctx) -> -spec kill(context()) -> ok | {error, not_isolated}. kill(Ctx) when is_pid(Ctx) -> case lookup_nif_ref(Ctx) of - {ok, isolated} -> + {ok, Marker} when is_atom(Marker) -> MRef = erlang:monitor(process, Ctx), Ctx ! {kill, self(), MRef}, await_ctrl_reply(Ctx, MRef, 5000); @@ -477,7 +477,7 @@ kill(Ctx) when is_pid(Ctx) -> -spec pass_fd(context(), non_neg_integer()) -> {ok, non_neg_integer()} | {error, term()}. pass_fd(Ctx, Fd) when is_pid(Ctx), is_integer(Fd) -> case lookup_nif_ref(Ctx) of - {ok, isolated} -> + {ok, Marker} when is_atom(Marker) -> MRef = erlang:monitor(process, Ctx), Ctx ! {pass_fd, self(), MRef, Fd}, await_ctrl_reply(Ctx, MRef, 5000); @@ -490,7 +490,7 @@ pass_fd(Ctx, Fd) when is_pid(Ctx), is_integer(Fd) -> -spec child_info(context()) -> {ok, map()} | {error, term()}. child_info(Ctx) when is_pid(Ctx) -> case lookup_nif_ref(Ctx) of - {ok, isolated} -> + {ok, Marker} when is_atom(Marker) -> MRef = erlang:monitor(process, Ctx), Ctx ! {child_info, self(), MRef}, await_ctrl_reply(Ctx, MRef, 5000); @@ -504,7 +504,7 @@ child_info(Ctx) when is_pid(Ctx) -> %% only interrupt whatever runs. interrupt_request(Ctx, ReqMRef) -> case lookup_nif_ref(Ctx) of - {ok, isolated} -> + {ok, Marker} when is_atom(Marker) -> Ctx ! {interrupt_request, ReqMRef}, ok; _ -> @@ -597,7 +597,7 @@ submit(Ctx, Module, Func, Args) -> {ok, reference()} | {error, term()}. submit(Ctx, Module, Func, Args, Kwargs) when is_pid(Ctx), is_list(Args), is_map(Kwargs) -> case lookup_nif_ref(Ctx) of - {ok, isolated} -> + {ok, Marker} when is_atom(Marker) -> TaskRef = make_ref(), MRef = erlang:monitor(process, Ctx), Ctx ! {submit, self(), MRef, TaskRef, to_binary(Module), to_binary(Func), Args, Kwargs}, @@ -688,7 +688,10 @@ await_ctrl_reply(Ctx, MRef, Timeout) -> {error, timeout} end. -%% @private +%% @private A context process registers either its NIF reference or an +%% atom (`isolated', `reimport_session'): an atom means "not a NIF +%% context, send it a message" for interrupt, kill, pass_fd, child_info and +%% submit. register_nif_ref(Ref) -> try true = ets:insert(?REF_TAB, {self(), Ref}), From a1b8c753f0524bd4834769046f775672b7342b2f Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Wed, 30 Sep 2026 11:33:14 +0200 Subject: [PATCH 14/18] Keep a re-import session's modules across several calls py_session:new/1 on a reimport template returns a process that answers like a context: calls, eval and exec run against the session's own module dictionary and __main__ in one of the template's contexts, so its fresh state lives from one call to the next and no other session sees it. Replies are relayed through the session process. A timeout does not interrupt, since the context is shared with other sessions: the call finishes and its late reply is dropped. The modules are freed on close, when the owner crashes, or through the template's monitor if the session process is killed. --- docs/code-map.md | 1 + priv/_erlang_impl/_reimport.py | 107 +++++++++++++-- src/py_reimport_session.erl | 237 +++++++++++++++++++++++++++++++++ src/py_session.erl | 11 +- src/py_session_template.erl | 54 +++++++- test/coverage_audit.md | 1 + test/py_session_SUITE.erl | 225 ++++++++++++++++++++++++++++++- test/py_session_soak_SUITE.erl | 56 +++++++- test/py_test_reimport.py | 24 ++++ 9 files changed, 693 insertions(+), 23 deletions(-) create mode 100644 src/py_reimport_session.erl diff --git a/docs/code-map.md b/docs/code-map.md index 0677683..647994f 100644 --- a/docs/code-map.md +++ b/docs/code-map.md @@ -19,6 +19,7 @@ exercised by suites). Guides are in `docs/`, suites in `test/`. Start with | `py_child` | Helpers shared by the processes that drive a Python child: executable lookup, socket listen/accept, frames, port env, rlimit flags | live | isolated | `py_isolated_*_SUITE` | | `py_session` | Isolated sessions: a fresh child per session from a template (`template/1`, `new/1`, `close/1`, `run/4`, `refresh/1`), or a fresh module dictionary per run (`start => reimport`) | live | sessions | `py_session_SUITE` | | `py_session_template`, `py_session_sup` | A template: zygotes that fork sessions (`priv/py_zygote.py`) or warm spawned sessions; exit reports to the session contexts | live | sessions | `py_session_SUITE` | +| `py_reimport_session` | A multi-call session of a re-import template: a process answering like a context, relaying calls to its context under the session's module dictionary | live | sessions | `py_session_SUITE` | | `py_context_router` | Pools and scheduler-affinity routing | live | pools, context-affinity | `py_context_router_SUITE`, `py_pool_SUITE` | | `py_context_sup`, `py_context_init` | Supervisor of contexts; starts the default pool at boot | live | pools | (through the above) | | `py_nif` | Erlang stubs and docs for every NIF | live | api-reference | all | diff --git a/priv/_erlang_impl/_reimport.py b/priv/_erlang_impl/_reimport.py index 6195b30..1540943 100644 --- a/priv/_erlang_impl/_reimport.py +++ b/priv/_erlang_impl/_reimport.py @@ -34,6 +34,7 @@ """ import builtins +import contextlib import importlib import sys import threading @@ -141,27 +142,115 @@ def pass_through(self, name, fromlist, mods): if key.startswith(prefix) and key not in mods: mods[key] = mod + def new_modules(self): + """A module dictionary holding only what this sandbox shares.""" + return {k: m for k, m in list(_real_modules.items()) if self.shared(k)} + def run(self, module, func, args=(), kwargs=None): - prev = (getattr(_tls, 'modules', None), getattr(_tls, 'sandbox', None)) - _tls.sandbox = self - _tls.modules = {k: m for k, m in list(_real_modules.items()) if self.shared(k)} - try: + with _active(self, self.new_modules()): fn = getattr(importlib.import_module(module), func) return fn(*args, **(kwargs or {})) - finally: - _tls.modules, _tls.sandbox = prev + + +@contextlib.contextmanager +def _active(sandbox, modules): + """Make `modules` this thread's sys.modules for the duration; nested + uses (a callback calling back in) restore the outer one.""" + prev = (getattr(_tls, 'modules', None), getattr(_tls, 'sandbox', None)) + _tls.sandbox = sandbox + _tls.modules = modules + try: + yield + finally: + _tls.modules, _tls.sandbox = prev _sandboxes = {} -def run(passthrough, module, func, args, kwargs): - """Entry point called by py_session:run/5 on a reimport template.""" +def _sandbox(passthrough): key = tuple(sorted(_as_text(p) for p in passthrough)) sandbox = _sandboxes.get(key) if sandbox is None: sandbox = _sandboxes[key] = Sandbox(key) - return sandbox.run(_as_text(module), _as_text(func), list(args), dict(kwargs or {})) + return sandbox + + +def run(passthrough, module, func, args, kwargs): + """Entry point called by py_session:run/5 on a reimport template.""" + return _sandbox(passthrough).run(_as_text(module), _as_text(func), + list(args), dict(kwargs or {})) + + +# --------------------------------------------------------------------------- +# Sessions spanning several calls (py_session:new/1 on a reimport template). +# Each keeps its module dictionary and its __main__ namespace between calls, +# by session id, until session_close. +# --------------------------------------------------------------------------- + +class _Session: + __slots__ = ('sandbox', 'modules', 'globals') + + def __init__(self, sandbox): + self.sandbox = sandbox + self.modules = sandbox.new_modules() + self.globals = {'__name__': '__main__', '__builtins__': builtins} + + +_sessions = {} + + +def _session(sid): + try: + return _sessions[sid] + except KeyError: + raise RuntimeError('reimport session %r is closed' % (sid,)) from None + + +def session_open(sid, passthrough): + _sessions[sid] = _Session(_sandbox(passthrough)) + return True + + +def session_call(sid, module, func, args, kwargs): + s = _session(sid) + module, func = _as_text(module), _as_text(func) + with _active(s.sandbox, s.modules): + if module in ('__main__', ''): + try: + fn = s.globals[func] + except KeyError: + raise AttributeError("name '%s' is not defined in the session" % func) from None + else: + fn = getattr(importlib.import_module(module), func) + return fn(*list(args), **dict(kwargs or {})) + + +def session_eval(sid, code, local_vars): + """Evaluated in the session's namespace; like a context's eval, + assignments made by the expression are not kept.""" + s = _session(sid) + scope = dict(s.globals) + scope.update({_as_text(k): v for k, v in dict(local_vars or {}).items()}) + with _active(s.sandbox, s.modules): + return eval(compile(_as_text(code), '', 'eval'), s.globals, scope) + + +def session_exec(sid, code): + s = _session(sid) + with _active(s.sandbox, s.modules): + exec(compile(_as_text(code), '', 'exec'), s.globals) + return True + + +def session_close(sid): + """Idempotent: the session process and the template may both close.""" + _sessions.pop(sid, None) + return True + + +def session_count(): + return len(_sessions) def _as_text(v): diff --git a/src/py_reimport_session.erl b/src/py_reimport_session.erl new file mode 100644 index 0000000..e07e998 --- /dev/null +++ b/src/py_reimport_session.erl @@ -0,0 +1,237 @@ +%% Copyright 2026 Benoit Chesneau +%% +%% Licensed under the Apache License, Version 2.0 (the "License"); +%% you may not use this file except in compliance with the License. +%% You may obtain a copy of the License at +%% +%% http://www.apache.org/licenses/LICENSE-2.0 +%% +%% Unless required by applicable law or agreed to in writing, software +%% distributed under the License is distributed on an "AS IS" BASIS, +%% WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +%% See the License for the specific language governing permissions and +%% limitations under the License. + +%%% @doc A session of a re-import template: several calls against one fresh +%%% module dictionary, kept in a worker or owngil context. +%%% +%%% The process is the handle py_session:new/1 returns. It answers the +%%% messages py_context sends to a context, so py_context:call/eval/exec, +%%% py:call(S, ...), timeouts and py_session:close/1 work on it. It never +%%% waits on Python: each call is sent to the context under a tag of its +%%% own and the reply is passed on to the caller when it comes. Nested calls +%%% (a callback calling back into the session) reach the context the same +%%% way and are served while it waits on the callback. +%%% +%%% A timeout does not interrupt: sessions share their context, and an +%%% embedded interrupt stops whatever the context runs, which may be another +%%% session's call. The caller stops waiting, the call finishes on its own +%%% and its late reply is dropped here. +%%% +%%% The session's modules and `__main__' live in the context +%%% (priv/_erlang_impl/_reimport.py, by session id) until close/1, the +%%% owner's crash, or, if this process is killed, the template's monitor. +%%% If the context dies the session answers `{error, {context_died, R}}' +%%% until it is closed. +%%% +%%% @private +%%% +%%% Owns: one session id in one context. +%%% Talks to: its context (forwarded requests), its template (registration). +%%% Never: runs Python itself or waits for a forwarded call. +-module(py_reimport_session). + +-behaviour(gen_server). + +-export([start_link/4]). +-export([init/1, handle_call/3, handle_cast/2, handle_info/2, terminate/2]). + +-define(REF_TAB, py_context_refs). +-define(IMPL, <<"_erlang_impl._reimport">>). +%% get_interp_id of a session: apart from real interpreter ids +-define(INTERP_ID_BASE, (1 bsl 48)). + +-record(st, { + ctx :: pid(), + ctx_mon :: reference(), + sid :: pos_integer(), + mode :: worker | owngil, + parent :: pid(), + %% calls sent to the context: Tag => {From, MRef, Kind} + pending = #{} :: #{reference() => {pid(), reference(), call | exec}}, + %% set once the context is gone + exited :: term() +}). + +%% @doc Open a session on Ctx, linked to the caller. +-spec start_link(pid(), pid(), [binary()], worker | owngil) -> {ok, pid()} | {error, term()}. +start_link(Template, Ctx, Passthrough, Mode) -> + Parent = self(), + Pid = proc_lib:spawn_link(fun() -> init_it(Parent, Template, Ctx, Passthrough, Mode) end), + MRef = erlang:monitor(process, Pid), + receive + {Pid, started} -> + erlang:demonitor(MRef, [flush]), + {ok, Pid}; + {Pid, {error, Reason}} -> + erlang:demonitor(MRef, [flush]), + unlink(Pid), + {error, Reason}; + {'DOWN', MRef, process, Pid, Reason} -> + unlink(Pid), + {error, Reason} + end. + +init_it(Parent, Template, Ctx, Passthrough, Mode) -> + process_flag(trap_exit, true), + %% Outlive a creator that exits normally, as other contexts do; its + %% crash stops the session (EXIT clause below) + put('$ancestors', [self() | case get('$ancestors') of + L when is_list(L) -> L; + _ -> [] + end]), + Sid = erlang:unique_integer([positive]), + case py_context:call(Ctx, ?IMPL, session_open, [Sid, Passthrough]) of + {ok, true} -> + ets:insert(?REF_TAB, {self(), reimport_session}), + Template ! {register_session, self(), Ctx, Sid}, + St = #st{ctx = Ctx, ctx_mon = erlang:monitor(process, Ctx), sid = Sid, + mode = Mode, parent = Parent}, + Parent ! {self(), started}, + gen_server:enter_loop(?MODULE, [], St); + {error, Reason} -> + Parent ! {self(), {error, {session_open_failed, Reason}}} + end. + +%% Not used: started through start_link/4 and enter_loop +init(_) -> + {stop, use_start_link}. + +handle_call(_Msg, _From, St) -> + {reply, {error, unknown_request}, St}. + +handle_cast(_Msg, St) -> + {noreply, St}. + +%% ---- requests, forwarded --------------------------------------------------- + +handle_info({call, From, MRef, M, F, A, K}, St) -> + forward(From, MRef, session_call, [M, F, A, K], St); +handle_info({call, From, MRef, M, F, A, K, _EnvRef}, St) -> + forward(From, MRef, session_call, [M, F, A, K], St); +handle_info({eval, From, MRef, Code, Locals}, St) -> + forward(From, MRef, session_eval, [Code, Locals], St); +handle_info({eval, From, MRef, Code, Locals, _EnvRef}, St) -> + forward(From, MRef, session_eval, [Code, Locals], St); +handle_info({exec, From, MRef, Code}, St) -> + forward(From, MRef, session_exec, [Code], St); +handle_info({exec, From, MRef, Code, _EnvRef}, St) -> + forward(From, MRef, session_exec, [Code], St); +handle_info({Tag, Reply}, #st{pending = Pending} = St) when is_map_key(Tag, Pending) -> + {{From, MRef, Kind}, Rest} = maps:take(Tag, Pending), + From ! {MRef, reply(Kind, Reply)}, + {noreply, St#st{pending = Rest}}; + +%% ---- introspection --------------------------------------------------------- + +handle_info({get_interp_id, From, MRef}, #st{sid = Sid} = St) -> + %% py:get_local_env/1 keys environments by this: one per session + From ! {MRef, {ok, ?INTERP_ID_BASE + Sid}}, + {noreply, St}; +handle_info({is_subinterp, From, MRef}, #st{mode = Mode} = St) -> + From ! {MRef, Mode =:= owngil}, + {noreply, St}; +handle_info({create_local_env, From, MRef}, St) -> + %% The session's namespace is the environment: the ref is not used + From ! {MRef, {ok, make_ref()}}, + {noreply, St}; +handle_info({child_info, From, MRef}, #st{ctx = Ctx, sid = Sid, mode = Mode} = St) -> + From ! {MRef, {ok, #{context => Ctx, session => Sid, mode => Mode}}}, + {noreply, St}; + +%% ---- controls -------------------------------------------------------------- + +handle_info({interrupt, From, MRef}, #st{exited = undefined, ctx = Ctx} = St) -> + %% An embedded interrupt stops what the context runs now + From ! {MRef, py_context:interrupt(Ctx)}, + {noreply, St}; +handle_info({interrupt, From, MRef}, St) -> + From ! {MRef, not_running}, + {noreply, St}; +handle_info({interrupt_request, ReqMRef}, #st{pending = Pending} = St) -> + %% The caller timed out. No interrupt (it could stop another session's + %% call): forget the request so its late reply is dropped + Drop = [Tag || {Tag, {_, M, _}} <- maps:to_list(Pending), M =:= ReqMRef], + {noreply, St#st{pending = maps:without(Drop, Pending)}}; +handle_info({kill, From, MRef}, St) -> + From ! {MRef, {error, not_supported}}, + {noreply, St}; +handle_info({Op, From, MRef, _}, St) + when Op =:= pass_fd; Op =:= start_loop -> + From ! {MRef, {error, not_supported_in_reimport}}, + {noreply, St}; +handle_info({stop_loop, From, MRef, _GraceMs}, St) -> + From ! {MRef, {error, no_loop}}, + {noreply, St}; +handle_info({Op, From, MRef}, St) when Op =:= get_nif_ref; Op =:= loop_ref -> + From ! {MRef, {error, not_supported_in_reimport}}, + {noreply, St}; +handle_info({call_method, From, MRef, _, _, _}, St) -> + From ! {MRef, {error, not_supported_in_reimport}}, + {noreply, St}; +handle_info({submit, From, MRef, _TaskRef, _M, _F, _A, _K}, St) -> + From ! {MRef, {error, not_supported_in_reimport}}, + {noreply, St}; +handle_info({cancel_ctrl, _MRef}, St) -> + {noreply, St}; + +%% ---- lifecycle ------------------------------------------------------------- + +handle_info({stop, From, MRef}, St) -> + close_session(St), + From ! {MRef, ok}, + {stop, normal, St#st{exited = closed}}; +handle_info({'DOWN', Mon, process, _Ctx, Reason}, #st{ctx_mon = Mon, pending = Pending} = St) -> + %% The context is gone and the session's modules with it: answer with + %% that until closed (stopping would take a linked caller down) + Exited = {context_died, Reason}, + [From ! {MRef, {error, Exited}} || {From, MRef, _} <- maps:values(Pending)], + {noreply, St#st{exited = Exited, pending = #{}}}; +handle_info({'EXIT', Parent, Reason}, #st{parent = Parent} = St) when Reason =/= normal -> + {stop, Reason, St}; +handle_info({'EXIT', _Pid, Reason}, St) + when Reason =:= shutdown; Reason =:= kill -> + {stop, Reason, St}; +handle_info(_Other, St) -> + {noreply, St}. + +terminate(_Reason, St) -> + close_session(St), + try ets:delete(?REF_TAB, self()) catch error:badarg -> ok end, + ok. + +%% ============================================================================ +%% Internal +%% ============================================================================ + +forward(From, MRef, _Fun, _Args, #st{exited = Exited} = St) when Exited =/= undefined -> + From ! {MRef, {error, Exited}}, + {noreply, St}; +forward(From, MRef, Fun, Args, #st{ctx = Ctx, sid = Sid, pending = Pending} = St) -> + %% The context replies {Tag, Result} to this process, which does not + %% wait for it + Tag = make_ref(), + Ctx ! {call, self(), Tag, ?IMPL, atom_to_binary(Fun), [Sid | Args], #{}}, + Kind = case Fun of session_exec -> exec; _ -> call end, + {noreply, St#st{pending = Pending#{Tag => {From, MRef, Kind}}}}. + +%% py_context:exec/2 returns ok, not {ok, _} +reply(exec, {ok, _}) -> ok; +reply(_Kind, Reply) -> Reply. + +%% Idempotent on the Python side; nothing to do once the context is gone +close_session(#st{exited = undefined, ctx = Ctx, sid = Sid}) -> + _ = py_context:call(Ctx, ?IMPL, session_close, [Sid], #{}, 5000), + ok; +close_session(_St) -> + ok. diff --git a/src/py_session.erl b/src/py_session.erl index 7b22a5f..c37b8b2 100644 --- a/src/py_session.erl +++ b/src/py_session.erl @@ -42,7 +42,8 @@ %%% again in a fresh module dictionary on one of the template's %%% `contexts' (`mode => worker | owngil'), as Temporal's workflow %%% sandbox does. Modules in `imports' and `passthrough', and the -%%% standard library, are shared by every run. +%%% standard library, are shared by every run. new/1 gives a session +%%% over several calls that keeps its fresh modules between them. %%%
  • `python', `paths', `imports', `preload' - the prepared state.
  • %%%
  • `env' - the environment of every session. Nothing is inherited %%% from the VM unless `clear_env => false'.
  • @@ -105,9 +106,11 @@ new(T, Opts) -> end; {start, CtxOpts} -> py_context:new(maps:merge(CtxOpts, maps:with([start_timeout], Opts))); - {reimport, _, _} -> - %% a re-import run is not a process: use run/5 - {error, {not_supported, reimport}}; + {reimport, Ctx, Passthrough} -> + %% A session over several calls: its own module dictionary in + %% that context, behind a process that answers like a context + py_reimport_session:start_link(T, Ctx, Passthrough, + py_session_template:mode(T)); {error, _} = Err -> Err end. diff --git a/src/py_session_template.erl b/src/py_session_template.erl index c30d0c5..f374b55 100644 --- a/src/py_session_template.erl +++ b/src/py_session_template.erl @@ -44,7 +44,8 @@ fork/4, checkout/2, refresh/1, - info/1]). + info/1, + mode/1]). -export([init/1, handle_call/3, handle_cast/2, handle_info/2, terminate/2]). @@ -71,6 +72,8 @@ %% start => reimport: the contexts runs go to, and what they share contexts = [] :: [pid()], passthrough = [] :: [binary()], + %% start => reimport: live multi-call sessions, Monitor => {Pid, Ctx, Sid} + rsessions = #{} :: #{reference() => {pid(), pid(), pos_integer()}}, zygotes = [] :: [#zygote{}], next = 0 :: non_neg_integer(), next_id = 1 :: pos_integer(), @@ -123,6 +126,11 @@ refresh(T) -> info(T) -> gen_server:call(T, info). +%% @doc `worker' or `owngil' for a reimport template. +-spec mode(pid()) -> worker | owngil. +mode(T) -> + gen_server:call(T, mode). + %% ============================================================================ %% gen_server callbacks %% ============================================================================ @@ -178,8 +186,9 @@ handle_call(refresh, _From, #st{start = fork, zygotes = Old} = St) -> end; handle_call(refresh, _From, #st{start = reimport, contexts = Old} = St) -> case build(St#st{contexts = []}) of - {ok, St1} -> + {ok, #st{contexts = [New | _]} = St1} -> [begin unlink(C), py_context:stop(C) end || C <- Old], + close_orphans(Old, New, St1), {reply, ok, St1}; {error, Reason} -> {reply, {error, Reason}, St} @@ -188,6 +197,8 @@ handle_call(refresh, _From, #st{start = spawn, warm = Warm} = St) -> [begin unlink(C), py_context:stop(C) end || C <- Warm], self() ! fill_warm, {reply, ok, St#st{warm = []}}; +handle_call(mode, _From, #st{opts = Opts} = St) -> + {reply, maps:get(mode, Opts, worker), St}; handle_call(info, _From, St) -> {reply, info_map(St), St}; handle_call(_Other, _From, St) -> @@ -196,6 +207,17 @@ handle_call(_Other, _From, St) -> handle_cast(_Msg, St) -> {noreply, St}. +handle_info({register_session, Pid, Ctx, Sid}, #st{rsessions = RS} = St) -> + %% Watched so a session process killed without closing still frees + %% its modules in the context + {noreply, St#st{rsessions = RS#{erlang:monitor(process, Pid) => {Pid, Ctx, Sid}}}}; +handle_info({'DOWN', Mon, process, _Pid, _Reason}, #st{rsessions = RS} = St) + when is_map_key(Mon, RS) -> + {{_, Ctx, Sid}, Rest} = maps:take(Mon, RS), + _ = spawn(fun() -> + py_context:call(Ctx, <<"_erlang_impl._reimport">>, session_close, [Sid], #{}, 5000) + end), + {noreply, St#st{rsessions = Rest}}; handle_info({'$socket', S, select, _}, St) -> case lists:keyfind(S, #zygote.sock, St#st.zygotes) of false -> {noreply, St}; @@ -239,6 +261,7 @@ handle_info({'EXIT', Pid, Reason}, #st{start = reimport, contexts = Cs} = St) -> [self(), Pid, Reason]), case start_context(St#st.opts) of {ok, C} -> + close_orphans([Pid], C, St), {noreply, St#st{contexts = [C | lists:delete(Pid, Cs)]}}; {error, Why} -> {stop, {context_restart_failed, Why}, St} @@ -252,7 +275,11 @@ handle_info({'EXIT', Pid, _Reason}, #st{warm = Warm} = St) -> handle_info(_Msg, St) -> {noreply, St}. -terminate(_Reason, #st{zygotes = Zs, warm = Warm, contexts = Cs}) -> +terminate(_Reason, #st{zygotes = Zs, warm = Warm, contexts = Cs, rsessions = RS}) -> + %% Worker contexts share the main interpreter, which outlives them: + %% free the modules of sessions still open before the contexts go + [py_context:call(Ctx, <<"_erlang_impl._reimport">>, session_close, [Sid], #{}, 5000) + || {_, Ctx, Sid} <- maps:values(RS), is_process_alive(Ctx)], [retire(Z) || Z <- Zs], [py_context:stop(C) || C <- Warm ++ Cs], ok. @@ -374,6 +401,23 @@ start_zygote(Opts) -> {error, {template_failed, Reason}} end. +%% The sessions of contexts that are gone. On worker contexts their modules +%% live in the main interpreter, which is still there: free them through +%% another context. On owngil they went with the interpreter. +close_orphans(Dead, Via, #st{opts = Opts, rsessions = RS}) -> + case maps:get(mode, Opts, worker) of + worker -> + Sids = [Sid || {_, Ctx, Sid} <- maps:values(RS), lists:member(Ctx, Dead)], + _ = Sids =/= [] andalso + spawn(fun() -> + [py_context:call(Via, <<"_erlang_impl._reimport">>, session_close, + [Sid], #{}, 5000) || Sid <- Sids] + end), + ok; + owngil -> + ok + end. + %% A context for re-import runs: the paths, the imports (shared by every %% run) and the preload are applied once. start_context(Opts) -> @@ -582,9 +626,11 @@ fork_opts(ForkOpts) -> end, #{}, ForkOpts). info_map(#st{start = Start, zygotes = Zs, owners = Owners, forks = Forks, - build_ms = BuildMs, warm = Warm, opts = Opts, contexts = Cs}) -> + build_ms = BuildMs, warm = Warm, opts = Opts, contexts = Cs, + rsessions = RS}) -> #{start => Start, contexts => Cs, + reimport_sessions => map_size(RS), zygotes => [Info#{os_pid => P} || #zygote{os_pid = P, info = Info} <- Zs], sessions => map_size(Owners), forks => Forks, diff --git a/test/coverage_audit.md b/test/coverage_audit.md index 25076ee..5087787 100644 --- a/test/coverage_audit.md +++ b/test/coverage_audit.md @@ -23,6 +23,7 @@ suite is visible. `scripts/check_code_map.sh` requires a row per module. | `py_session` | `py_session_SUITE`, `py_session_stress_SUITE`, `py_session_soak_SUITE` | | `py_session_template` | `py_session_SUITE`, `py_session_stress_SUITE`, `py_session_soak_SUITE` | | `py_session_sup` | (through `py_session_SUITE`) | +| `py_reimport_session` | `py_session_SUITE`, `py_session_soak_SUITE` | | `py_context_router` | `py_context_router_SUITE`, `py_pool_SUITE` | | `py_context_sup` | (through the above) | | `py_context_init` | (through the above) | diff --git a/test/py_session_SUITE.erl b/test/py_session_SUITE.erl index 458f37f..7671fec 100644 --- a/test/py_session_SUITE.erl +++ b/test/py_session_SUITE.erl @@ -46,7 +46,17 @@ test_reimport_reentrance/1, test_reimport_typing_and_pickle/1, test_reimport_errors/1, - test_reimport_no_session_process/1, + test_rsession_state_across_calls/1, + test_rsession_exec_eval/1, + test_rsession_py_call/1, + test_rsession_reentrance/1, + test_rsession_concurrent/1, + test_rsession_cleanup/1, + test_rsession_timeout/1, + test_rsession_context_died/1, + test_rsession_unsupported/1, + test_rsession_threads_see_context/1, + test_rsession_template_stop_frees/1, test_reimport_refresh/1, test_reimport_bad_options/1 ]). @@ -86,7 +96,17 @@ groups() -> test_reimport_reentrance, test_reimport_typing_and_pickle, test_reimport_errors, - test_reimport_no_session_process, + test_rsession_state_across_calls, + test_rsession_exec_eval, + test_rsession_py_call, + test_rsession_reentrance, + test_rsession_concurrent, + test_rsession_cleanup, + test_rsession_timeout, + test_rsession_context_died, + test_rsession_unsupported, + test_rsession_threads_see_context, + test_rsession_template_stop_frees, test_reimport_refresh, test_reimport_bad_options ], @@ -617,11 +637,208 @@ test_reimport_errors(Config) -> {ok, 1} = py_session:run(T, ?RMOD, bump, []), ok. -test_reimport_no_session_process(Config) -> +%% A session keeps its fresh module state between calls; another session +%% and run/5 start from scratch. +test_rsession_state_across_calls(Config) -> T = rtemplate(Config), - {error, {not_supported, reimport}} = py_session:new(T), + {ok, S} = py_session:new(T), + [{ok, N} = py_context:call(S, ?RMOD, bump, []) || N <- [1, 2, 3]], + {ok, S2} = py_session:new(T), + {ok, 1} = py_context:call(S2, ?RMOD, bump, []), + {ok, 1} = py_session:run(T, ?RMOD, bump, []), + {ok, 4} = py_context:call(S, ?RMOD, bump, []), + py_session:close(S), + py_session:close(S2). + +%% exec and eval use the session's own __main__. +test_rsession_exec_eval(Config) -> + T = rtemplate(Config), + {ok, A} = py_session:new(T), + {ok, B} = py_session:new(T), + ok = py_context:exec(A, <<"x = 21\ndef twice(v): return v * 2">>), + {ok, 42} = py_context:eval(A, <<"x * 2">>), + {ok, 10} = py_context:call(A, '__main__', twice, [5]), + {ok, 7} = py_context:eval(A, <<"x - y">>, #{y => 14}), + {error, {'NameError', _}} = py_context:eval(B, <<"x">>), + py_session:close(A), + py_session:close(B). + +%% py:call/eval on a session go through the process-local env path. +test_rsession_py_call(Config) -> + T = rtemplate(Config), + {ok, S} = py_session:new(T), + {ok, 1} = py:call(S, ?RMOD, bump, []), + {ok, 2} = py:call(S, ?RMOD, bump, []), + py_session:close(S). + +%% A callback calls back into the same session and sees the same module +%% state; a callback of session A calls into session B. +test_rsession_reentrance(Config) -> + T = rtemplate(Config, #{contexts => 1}), + py:register_function(rsess_cb, fun([S]) -> + {ok, N} = py_context:call(S, ?RMOD, bump, [], #{}, 10000), + N + end), + try + {ok, A} = py_session:new(T), + {ok, B} = py_session:new(T), + {ok, 1} = py_context:call(A, ?RMOD, bump, []), + %% the nested bump runs in A: 2, and the outer call then sees 2 + {ok, {2, 2}} = py_context:call(A, ?RMOD, bump_via_callback, [A], #{}, 10000), + %% A's callback bumps B: B's count moves, A's does not + {ok, {1, 2}} = py_context:call(A, ?RMOD, bump_via_callback, [B], #{}, 10000), + {ok, 2} = py_context:call(B, ?RMOD, bump, []), + py_session:close(A), + py_session:close(B) + after + py:unregister_function(rsess_cb) + end. + +test_rsession_concurrent(Config) -> + T = rtemplate(Config, #{contexts => 2}), + Self = self(), + Pids = [spawn_link(fun() -> + {ok, S} = py_session:new(T), + R = [py_context:call(S, ?RMOD, bump, []) || _ <- lists:seq(1, 20)], + py_session:close(S), + Self ! {done, R} + end) || _ <- lists:seq(1, 16)], + Want = [{ok, N} || N <- lists:seq(1, 20)], + [receive {done, R} -> R = Want after 60000 -> ct:fail(timeout) end || _ <- Pids], + 0 = wait_rsessions(T, 0, 100), ok. +%% The session's modules are freed on close, when the owner crashes, and +%% when the session process is killed (the template's monitor). +test_rsession_cleanup(Config) -> + T = rtemplate(Config, #{contexts => 1}), + #{contexts := [Ctx]} = py_session:info(T), + Count = fun() -> {ok, N} = py_context:call(Ctx, <<"_erlang_impl._reimport">>, session_count, []), N end, + 0 = Count(), + {ok, S} = py_session:new(T), + 1 = Count(), + ok = py_session:close(S), + 0 = Count(), + Self = self(), + {Owner, OMon} = spawn_monitor(fun() -> + {ok, S1} = py_session:new(T), + Self ! {session, S1}, + receive never -> ok end + end), + S1 = receive {session, X} -> X after 10000 -> ct:fail(no_session) end, + SMon = erlang:monitor(process, S1), + exit(Owner, crash), + receive {'DOWN', OMon, process, Owner, _} -> ok end, + receive {'DOWN', SMon, process, S1, _} -> ok after 5000 -> ct:fail(session_survived) end, + ok = wait_count(Count, 0, 100), + {ok, S2} = py_session:new(T), + unlink(S2), + exit(S2, kill), + ok = wait_count(Count, 0, 100), + 0 = wait_rsessions(T, 0, 100), + ok. + +%% A timeout does not interrupt (it could hit another session sharing the +%% context): the caller stops waiting, the call finishes, its late reply is +%% dropped, and the session goes on. +test_rsession_timeout(Config) -> + T = rtemplate(Config, #{contexts => 1}), + {ok, S} = py_session:new(T), + {ok, Other} = py_session:new(T), + {ok, 1} = py_context:call(S, ?RMOD, bump, []), + {error, timeout} = py_context:call(S, ?RMOD, sleep, [1], #{}, 200), + %% queued behind the sleep, and not interrupted by the timeout + {ok, 1} = py_context:call(Other, ?RMOD, bump, [], #{}, 5000), + {ok, 2} = py_context:call(S, ?RMOD, bump, []), + %% no late reply left in the caller's mailbox + receive {_, {ok, <<"slept">>}} -> ct:fail(late_reply_delivered) after 200 -> ok end, + py_session:close(S), + py_session:close(Other). + +%% The context under a session goes: the session answers with that until +%% closed, and does not take its linked owner down. +test_rsession_context_died(Config) -> + T = rtemplate(Config, #{contexts => 1}), + #{contexts := [Ctx]} = py_session:info(T), + Before = session_count(Ctx), + {ok, S} = py_session:new(T), + {ok, 1} = py_context:call(S, ?RMOD, bump, []), + ok = py_context:stop(Ctx), + {error, {context_died, _}} = wait_context_died(S, 50), + {error, {context_died, _}} = py_context:exec(S, <<"x = 1">>), + true = is_process_alive(S), + ok = py_session:close(S), + false = is_process_alive(S), + %% its modules are freed: through the replacing context on worker (the + %% main interpreter outlives the context), with the interpreter on owngil + #{contexts := [New]} = py_session:info(T), + Expected = case ?config(mode, Config) of worker -> Before; owngil -> 0 end, + ok = wait_count(fun() -> session_count(New) end, Expected, 100), + ok. + +%% Stopping a template frees the modules of its sessions still open. +test_rsession_template_stop_frees(Config) -> + case ?config(mode, Config) of + owngil -> + {skip, "owngil sessions go with their interpreter"}; + worker -> + {ok, Probe} = py_context:new(#{mode => worker}), + Before = session_count(Probe), + T = rtemplate(Config), + {ok, S} = py_session:new(T), + unlink(S), + {ok, 1} = py_context:call(S, ?RMOD, bump, []), + Before1 = Before + 1, + Before1 = session_count(Probe), + py_session:stop_template(T), + ok = wait_count(fun() -> session_count(Probe) end, Before, 100), + py_context:stop(Probe) + end. + +session_count(Ctx) -> + {ok, N} = py_context:call(Ctx, <<"_erlang_impl._reimport">>, session_count, []), + N. + +test_rsession_unsupported(Config) -> + T = rtemplate(Config), + {ok, S} = py_session:new(T), + {error, not_supported} = py_context:kill(S), + {error, not_supported_in_reimport} = py_context:pass_fd(S, 0), + {error, not_supported_in_reimport} = py_context:start_loop(S), + {ok, #{session := _, context := _}} = py_context:child_info(S), + py_session:close(S). + +%% Threads started in a session see the context's modules (documented). +test_rsession_threads_see_context(Config) -> + T = rtemplate(Config), + {ok, S} = py_session:new(T), + {ok, {true, false}} = py_context:call(S, ?RMOD, thread_sees, []), + py_session:close(S). + +wait_context_died(S, 0) -> + py_context:call(S, ?RMOD, bump, []); +wait_context_died(S, N) -> + case py_context:call(S, ?RMOD, bump, []) of + {error, {context_died, _}} = E -> E; + _ -> timer:sleep(20), wait_context_died(S, N - 1) + end. + +wait_count(_Count, _Want, 0) -> + ct:fail(session_not_freed); +wait_count(Count, Want, N) -> + case Count() of + Want -> ok; + _ -> timer:sleep(20), wait_count(Count, Want, N - 1) + end. + +wait_rsessions(T, Want, 0) -> + maps:get(reimport_sessions, py_session:info(T), Want); +wait_rsessions(T, Want, N) -> + case py_session:info(T) of + #{reimport_sessions := Want} -> Want; + _ -> timer:sleep(20), wait_rsessions(T, Want, N - 1) + end. + test_reimport_refresh(Config) -> T = rtemplate(Config, #{contexts => 2}), #{contexts := Old} = py_session:info(T), diff --git a/test/py_session_soak_SUITE.erl b/test/py_session_soak_SUITE.erl index 665436e..2e3b00a 100644 --- a/test/py_session_soak_SUITE.erl +++ b/test/py_session_soak_SUITE.erl @@ -39,6 +39,8 @@ test_session_churn_no_leak(Config) -> imports => [?MOD]}), {ok, Spawn} = py_session:template(#{start => spawn, warm => 2, paths => [TestDir], imports => [?MOD]}), + {ok, Reimport} = py_session:template(#{start => reimport, mode => worker, contexts => 2, + paths => [TestDir]}), timer:sleep(1000), Base = counters(), ct:print("baseline: ~p", [Base]), @@ -46,8 +48,11 @@ test_session_churn_no_leak(Config) -> Self = self(), Workers = [spawn_link(fun() -> rand:seed(exsss, {I, I * 7, I * 13}), - T = case I rem 4 of 0 -> Spawn; _ -> Fork end, - Self ! {done, self(), churn(T, Deadline, 0, [])} + Self ! {done, self(), case I rem 4 of + 0 -> churn(Spawn, Deadline, 0, []); + 1 -> rchurn(Reimport, Deadline, 0, []); + _ -> churn(Fork, Deadline, 0, []) + end} end) || I <- lists:seq(1, 12)], Results = [receive {done, W, R} -> R after (Seconds + 120) * 1000 -> ct:fail(worker_hung) end || W <- Workers], @@ -58,6 +63,10 @@ test_session_churn_no_leak(Config) -> [] = Errs, true = Ops > 0, #{sessions := 0} = wait_idle(Fork, 100), + %% every re-import session freed its modules in its context + #{contexts := RCtxs} = py_session:info(Reimport), + [0 = wait_freed(C, 100) || C <- RCtxs], + py_session:stop_template(Reimport), py_session:stop_template(Fork), py_session:stop_template(Spawn), timer:sleep(1000), @@ -77,6 +86,49 @@ churn(T, Deadline, N, Errs) -> churn(T, Deadline, N + 1, case Got of ok -> Errs; Bad -> [Bad | Errs] end) end. +%% A re-import session: several calls on one fresh state, then closed, +%% timed out, or abandoned by an owner that crashes +rchurn(T, Deadline, N, Errs) -> + case erlang:monotonic_time(millisecond) < Deadline of + false -> + {N, Errs}; + true -> + Got = case rand:uniform(3) of + 1 -> + {ok, S} = py_session:new(T), + R = [py_context:call(S, py_test_reimport, bump, []) || _ <- lists:seq(1, 3)], + ok = py_session:close(S), + expect([{ok, 1}, {ok, 2}, {ok, 3}], R); + 2 -> + {ok, S} = py_session:new(T), + R = py_context:call(S, py_test_reimport, sleep, [0.2], #{}, 50), + {ok, 1} = py_context:call(S, py_test_reimport, bump, []), + ok = py_session:close(S), + expect({error, timeout}, R); + 3 -> + Self = self(), + {Owner, Mon} = spawn_monitor(fun() -> + {ok, S} = py_session:new(T), + {ok, 1} = py_context:call(S, py_test_reimport, bump, []), + Self ! {owner_ready, self()}, + exit(crash) + end), + receive {owner_ready, Owner} -> ok after 10000 -> ok end, + receive {'DOWN', Mon, process, Owner, _} -> ok end, + ok + end, + rchurn(T, Deadline, N + 1, case Got of ok -> Errs; Bad -> [Bad | Errs] end) + end. + +wait_freed(Ctx, 0) -> + {ok, N} = py_context:call(Ctx, <<"_erlang_impl._reimport">>, session_count, []), + N; +wait_freed(Ctx, Tries) -> + case py_context:call(Ctx, <<"_erlang_impl._reimport">>, session_count, []) of + {ok, 0} -> 0; + _ -> timer:sleep(50), wait_freed(Ctx, Tries - 1) + end. + use(1, S) -> expect({ok, 3}, py_context:call(S, ?MOD, reenter, [3])); use(2, S) -> expect({error, {child_exited, {signal, 6}}}, py_context:call(S, ?MOD, abort, [])); use(3, S) -> expect(ok, py_context:kill(S)); diff --git a/test/py_test_reimport.py b/test/py_test_reimport.py index 66a506c..f33ccaa 100644 --- a/test/py_test_reimport.py +++ b/test/py_test_reimport.py @@ -67,3 +67,27 @@ def fail(msg): def whoami(): return __name__ + + +def sleep(seconds): + import time + time.sleep(seconds) + return 'slept' + + +def bump_via_callback(session): + """Calls Erlang, whose callback calls back into this same session.""" + from_callback = erlang.call('rsess_cb', session) + return (from_callback, COUNT['n']) + + +def thread_sees(): + """The module swap is per thread: a thread started here sees the + context's modules, where this module was never imported.""" + import sys + import threading + seen = {} + t = threading.Thread(target=lambda: seen.update(thread='py_test_reimport' in sys.modules)) + t.start() + t.join() + return ('py_test_reimport' in sys.modules, seen['thread']) From 6d5d89972ecbe0f7f20a3148dbe0870a93747ae1 Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Wed, 30 Sep 2026 11:34:40 +0200 Subject: [PATCH 15/18] Document re-import sessions over several calls How to open one and what it keeps, why a timeout does not interrupt there, and what a call inside costs; with a bench row for it. --- CHANGELOG.md | 6 ++- docs/decisions/0009-isolated-sessions.md | 4 ++ docs/sessions.md | 47 ++++++++++++++++++++---- examples/bench_sessions.erl | 8 +++- 4 files changed, 56 insertions(+), 9 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 096cb34..0af2fe5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -22,7 +22,11 @@ call in a fresh module dictionary on a worker or owngil context instead, the way Temporal's Python SDK isolates a workflow run: the function's module is imported again per run, and the standard library, `imports` and - `passthrough` modules are shared. See `docs/sessions.md`. + `passthrough` modules are shared. On such a template `new/1` gives a + session that keeps its fresh modules and `__main__` over several calls, + used like any context (`py_context:call/eval/exec`, `py:call`). A timeout + there does not interrupt, since the context is shared with other + sessions. See `docs/sessions.md`. - `clear_env` and `hash_seed` options for isolated contexts: the child sees only the variables named in `env`, and every child built with the same seed hashes strings and orders sets the same way. diff --git a/docs/decisions/0009-isolated-sessions.md b/docs/decisions/0009-isolated-sessions.md index 653043f..f0850f2 100644 --- a/docs/decisions/0009-isolated-sessions.md +++ b/docs/decisions/0009-isolated-sessions.md @@ -59,3 +59,7 @@ namespaces), and Wasm images (no native C extensions). interpreter's own module table (the C `pickle`) does not see a run's modules; the NIF's own lookups use `PyImport_GetModuleDict()` for that reason. +- A re-import session over several calls is a process that answers like a + context and relays calls to a context shared with other sessions. An + embedded interrupt cannot target one session's call, so its timeouts do + not interrupt: the call finishes and the late reply is dropped. diff --git a/docs/sessions.md b/docs/sessions.md index a67ee3f..ee1d6f0 100644 --- a/docs/sessions.md +++ b/docs/sessions.md @@ -95,9 +95,37 @@ SDK isolates a workflow run: Each `run/5` imports `orders` again in a new module dictionary, so its globals start fresh and nothing a run leaves in them reaches the next one. The standard library, `erlang`, `imports` and `passthrough` modules are -shared with the context. A run is one call: `py_session:new/1` answers -`{error, {not_supported, reimport}}`. Calls back into the same context from -a callback work. +shared with the context. Calls back into the same context from a callback +work. + +When the fresh state must last over several calls, open a session: + +```erlang +{ok, S} = py_session:new(T), +{ok, _} = py_context:call(S, orders, open, [OrderId]), +{ok, _} = py_context:call(S, orders, apply, [Event1]), +{ok, _} = py_context:call(S, orders, apply, [Event2]), +{ok, Total} = py_context:call(S, orders, total, []), +ok = py_session:close(S). +``` + +`orders` is imported fresh on the first call and kept for the next ones; +another session has its own copy. `py_context:exec/eval` use the session's +own `__main__`, and `py:call(S, ...)` works too. A callback can call back +into the same session. The modules are freed on `close/1`, when the +process that opened the session crashes, or if the session process is +killed. + +The session runs on one of the template's contexts, shared with other +sessions, so two things differ from a fork session: + +- A timeout does not interrupt the call: an interrupt would stop whatever + the context runs, maybe another session's call. You get + `{error, timeout}`, the call finishes on its own and its late reply is + dropped. `py_context:interrupt(S)` still interrupts the context's running + call; `py_context:kill(S)` answers `{error, not_supported}`. +- If the context goes away, the session answers + `{error, {context_died, Reason}}` until you close it. | | `fork` / `spawn` | `reimport` | |---|---|---| @@ -105,12 +133,16 @@ a callback work. | Standard library, `imports`, `passthrough` | fresh (`fork`: as prepared) | shared, their state persists | | C extension state, environment, working directory, threads, hash seed | fresh (hash seed per template) | shared with the context | | A call stuck in C | killed | runs on; interrupts land at the next bytecode | +| A timed-out call | interrupted, then killed after `kill_after` | finishes on its own; the caller stops waiting | | A segfault in a C extension | kills the session | kills the node | | Memory and CPU limits | yes | no | -| Cost per run | a fork, or a warm child | an import of the function's module | -| Cost per call inside the run | a local socket round trip | none | +| Cost per run or session | a fork, or a warm child | an import of the function's module | +| Cost per call inside the session | a local socket round trip | two Erlang messages | + +Three things to know about re-import runs and sessions: -Two things to know about re-import runs: +- Threads a session starts itself (and executor workers) see the context's + modules, not the session's: the module dictionary is swapped per thread. - `sys.modules` in that interpreter becomes a mapping that shows each thread its run's modules, and `builtins.__import__` is replaced, from the first @@ -205,7 +237,8 @@ One zygote forks one session at a time; add `zygotes` when sessions are opened faster than one zygote forks them. A call inside a session costs what it costs in any isolated context (30 us p50, 70 us for a call that calls back into the session), since it crosses the same socket. A call in -a re-import run stays in the process. +a re-import session stays in the process and costs about 20 us, the two +Erlang messages through the session process included. `examples/bench_sessions_sdks.py` measures the isolation step of Temporal's workflow sandbox (0.5 ms for a standard-library workflow, 4.6 ms when it diff --git a/examples/bench_sessions.erl b/examples/bench_sessions.erl index 5fb15bf..b3928a8 100644 --- a/examples/bench_sessions.erl +++ b/examples/bench_sessions.erl @@ -81,7 +81,13 @@ main(_) -> [us(fun() -> {ok, _} = py_session:run(R, bench_reimport_wf, handle, [[], #{}]) end) || _ <- lists:seq(1, 500)]), L = reimport_throughput(R, 8, 3), - io:format(" ~-18s ~7.1f runs/s with 8 callers~n", [M, length(L) / 3]) + io:format(" ~-18s ~7.1f runs/s with 8 callers~n", [M, length(L) / 3]), + {ok, RS} = py_session:new(R), + {ok, _} = py_context:call(RS, bench_reimport_wf, handle, [[], #{}]), + row(atom_to_list(M) ++ " session call", + [us(fun() -> {ok, _} = py_context:call(RS, bench_reimport_wf, handle, [[], #{}]) end) + || _ <- lists:seq(1, 2000)]), + py_session:close(RS) end || M <- [worker] ++ [owngil || py_nif:owngil_supported()]], ok. From 9f561361599d86bee8e2df0d6112281f79028f3b Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Wed, 30 Sep 2026 18:41:45 +0200 Subject: [PATCH 16/18] Release owngil process envs when their process exits The destructor of a process-local env runs on a thread that cannot take an owngil interpreter's GIL, so it dropped the dicts unreleased and they stayed until the context stopped: 300 processes leaving 1 MB each grew the node by 300 MB. The env now keeps its context and hands the dicts to it on a lock-free list; the context thread releases them before its next request and before the interpreter ends. The namespaces an owngil loop keeps per process had the same problem. They are created in the main interpreter by event_loop_exec/eval, so each records its interpreter and is released there when its process exits. --- CHANGELOG.md | 7 ++ c_src/py_event_loop.c | 65 +++++++++++++--- c_src/py_event_loop.h | 10 +++ c_src/py_nif.c | 111 +++++++++++++++++++++++----- c_src/py_nif.h | 29 ++++++++ docs/owngil_internals.md | 13 ++-- docs/process-bound-envs.md | 16 ++-- test/py_context_async_env_SUITE.erl | 109 ++++++++++++++++++++++++++- 8 files changed, 321 insertions(+), 39 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 0af2fe5..379a958 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -33,6 +33,13 @@ ### Fixed +- On an owngil context, the process-local environment of an Erlang process + (`py:call/eval/exec` on the context) was kept until the context stopped: + 300 short-lived processes each leaving 1 MB grew the node by 300 MB. The + environment now goes back to its context thread when the process exits + and is released before the context's next request. The per-process + namespaces of an owngil context's event loop (`py_event_loop:exec/eval`) + were not released either; they now are. - On an owngil context, a Python function that called `erlang.call`, where the Erlang callback called the same context again, hung until the request timeout. The context thread waited for the callback on its pipe while the diff --git a/c_src/py_event_loop.c b/c_src/py_event_loop.c index fe4080a..ac4ae95 100644 --- a/c_src/py_event_loop.c +++ b/c_src/py_event_loop.c @@ -529,17 +529,37 @@ void event_loop_destructor(ErlNifEnv *env, void *obj) { pthread_mutex_unlock(&loop->namespaces_mutex); PyGILState_Release(gstate); } else { - /* Subinterpreter or runtime not running: just free structs */ + /* Subinterpreter or runtime not running. Namespaces created in the + * subinterpreter went (or go) with it; those created in the main + * interpreter (event_loop_exec/eval) are released under the main + * GIL, taken after the lock is dropped. */ pthread_mutex_lock(&loop->namespaces_mutex); + process_namespace_t *ns_list = loop->namespaces_head; + loop->namespaces_head = NULL; + pthread_mutex_unlock(&loop->namespaces_mutex); - process_namespace_t *ns = loop->namespaces_head; + bool main_gil = runtime_is_running() && + PyGILState_GetThisThreadState() == NULL && !PyGILState_Check(); + PyGILState_STATE gstate = PyGILState_UNLOCKED; + if (main_gil) { + gstate = PyGILState_Ensure(); + } + process_namespace_t *ns = ns_list; while (ns != NULL) { process_namespace_t *next = ns->next; - /* Skip Py_XDECREF - can't safely acquire GIL */ + if (main_gil && ns->interp_id == 0) { + Py_XDECREF(ns->globals); + Py_XDECREF(ns->locals); + Py_XDECREF(ns->module_cache); + } enif_free(ns); ns = next; } - loop->namespaces_head = NULL; + if (main_gil) { + PyGILState_Release(gstate); + } + + pthread_mutex_lock(&loop->namespaces_mutex); /* Clean up PID-to-env mappings */ pid_env_mapping_t *mapping = loop->pid_env_head; @@ -693,11 +713,15 @@ void event_loop_down(ErlNifEnv *env, void *obj, ErlNifPid *pid, erlang_event_loop_t *loop = (erlang_event_loop_t *)obj; /* - * For subinterpreters (interp_id != 0), we can't use PyGILState_Ensure. - * Just remove from the list without Py_DECREF - the Python objects will - * be cleaned up when the interpreter is destroyed. + * Subinterpreter loop. Its namespaces are not all its own: event_loop_exec + * and event_loop_eval create theirs in the main interpreter. Unlink under + * the lock; hand subinterpreter dicts to the OWN_GIL context thread while + * the lock guarantees that context is still serving; release + * main-interpreter dicts under the main GIL once the lock is dropped + * (GIL before namespaces_mutex, as everywhere else). */ if (!runtime_is_running() || loop->interp_id != 0) { + process_namespace_t *main_ns = NULL; pthread_mutex_lock(&loop->namespaces_mutex); process_namespace_t **pp = &loop->namespaces_head; @@ -705,14 +729,33 @@ void event_loop_down(ErlNifEnv *env, void *obj, ErlNifPid *pid, if (enif_compare_pids(&(*pp)->owner_pid, pid) == 0) { process_namespace_t *to_free = *pp; *pp = to_free->next; - /* Skip Py_XDECREF - can't safely acquire GIL for subinterp */ - enif_free(to_free); + if (!runtime_is_running()) { + enif_free(to_free); + } else if (to_free->interp_id == 0) { + main_ns = to_free; + } else { + if (loop->gc_ctx != NULL) { + env_gc_push(loop->gc_ctx, to_free->globals, to_free->locals, + to_free->module_cache); + } + /* else: they go with the interpreter */ + enif_free(to_free); + } break; } pp = &(*pp)->next; } pthread_mutex_unlock(&loop->namespaces_mutex); + + if (main_ns != NULL) { + PyGILState_STATE gstate = PyGILState_Ensure(); + Py_XDECREF(main_ns->globals); + Py_XDECREF(main_ns->locals); + Py_XDECREF(main_ns->module_cache); + PyGILState_Release(gstate); + enif_free(main_ns); + } return; } @@ -812,6 +855,10 @@ static process_namespace_t *ensure_process_namespace( } ns->owner_pid = *pid; + { + PyInterpreterState *creator = PyInterpreterState_Get(); + ns->interp_id = creator != NULL ? PyInterpreterState_GetID(creator) : 0; + } ns->globals = PyDict_New(); ns->locals = PyDict_New(); ns->module_cache = PyDict_New(); diff --git a/c_src/py_event_loop.h b/c_src/py_event_loop.h index 52d890d..fce2fb1 100644 --- a/c_src/py_event_loop.h +++ b/c_src/py_event_loop.h @@ -121,6 +121,11 @@ typedef struct process_namespace { /** @brief Module import cache for this process */ PyObject *module_cache; + /** @brief Interpreter the dicts were created in (0 = main). Not always + * the loop's: event_loop_exec/eval run on the main interpreter even for + * a subinterpreter loop. */ + int64_t interp_id; + /** @brief Monitor for detecting process death */ ErlNifMonitor monitor; @@ -470,6 +475,11 @@ typedef struct erlang_event_loop { /** @brief Mutex protecting namespace registry */ pthread_mutex_t namespaces_mutex; + /** @brief OWN_GIL context whose thread releases the namespaces of dead + * processes (NULL for the main interpreter, and once the context stops + * serving). Protected by namespaces_mutex. */ + struct py_context *gc_ctx; + /* ========== PID-to-Env Mapping Registry ========== */ /* Protected by namespaces_mutex (shared with namespace registry) */ diff --git a/c_src/py_nif.c b/c_src/py_nif.c index 7ff4fa5..5406cae 100644 --- a/c_src/py_nif.c +++ b/c_src/py_nif.c @@ -105,36 +105,84 @@ static void py_env_resource_dtor(ErlNifEnv *env, void *obj) { (void)env; py_env_resource_t *res = (py_env_resource_t *)obj; - if (!runtime_is_running()) { + if (res->owner_ctx != NULL) { + /* OWN_GIL env: its dicts belong to the context's interpreter, whose + * GIL this thread cannot take. Hand them to the context thread, + * which releases them before its next request (env_gc_drain). */ + py_context_t *ctx = res->owner_ctx; + res->owner_ctx = NULL; + env_gc_push(ctx, res->globals, res->locals, NULL); res->globals = NULL; res->locals = NULL; + enif_release_resource(ctx); return; } - PyGILState_STATE gstate = PyGILState_Ensure(); - -#ifdef HAVE_SUBINTERPRETERS - if (res->interp_id != 0) { - /* OWN_GIL subinterpreter: interp_id != 0 - * These dicts were created in an OWN_GIL interpreter. We cannot safely - * DECREF them here because: - * 1. The interpreter might already be destroyed - * 2. We cannot switch to its thread state from this thread - * When the OWN_GIL context is destroyed, Py_EndInterpreter cleans up - * all objects, so we skip DECREF to avoid double-free or invalid access. */ - } else -#endif - { - /* Main interpreter */ - Py_XDECREF(res->globals); - Py_XDECREF(res->locals); + if (!runtime_is_running() || res->interp_id != 0) { + /* Runtime gone, or a subinterpreter env without an owner context: + * nothing can release it here */ + res->globals = NULL; + res->locals = NULL; + return; } + /* Main interpreter */ + PyGILState_STATE gstate = PyGILState_Ensure(); + Py_XDECREF(res->globals); + Py_XDECREF(res->locals); PyGILState_Release(gstate); res->globals = NULL; res->locals = NULL; } +void env_gc_push(py_context_t *ctx, PyObject *globals, PyObject *locals, PyObject *extra) { + if (globals == NULL && locals == NULL && extra == NULL) { + return; + } + env_gc_node_t *node = enif_alloc(sizeof(env_gc_node_t)); + if (node == NULL) { + return; /* the objects go with the interpreter */ + } + node->globals = globals; + node->locals = locals; + node->extra = extra; + node->next = atomic_load(&ctx->env_gc_head); + while (!atomic_compare_exchange_weak(&ctx->env_gc_head, &node->next, node)) { + /* node->next was reloaded: retry */ + } +} + +/** + * @brief Release the dicts of dead process-local envs of an OWN_GIL context. + * + * Runs on the context thread with its GIL held. Takes the whole pending list + * at once; the env destructor only ever pushes, so no lock is needed. + */ +static void env_gc_drain(py_context_t *ctx) { + env_gc_node_t *node = atomic_exchange(&ctx->env_gc_head, NULL); + while (node != NULL) { + env_gc_node_t *next = node->next; + Py_XDECREF(node->globals); + Py_XDECREF(node->locals); + Py_XDECREF(node->extra); + enif_free(node); + node = next; + } +} + +/** + * @brief Free the nodes left after the interpreter ended, without touching + * their objects (they went with the interpreter). Context destructor only. + */ +static void env_gc_discard(py_context_t *ctx) { + env_gc_node_t *node = atomic_exchange(&ctx->env_gc_head, NULL); + while (node != NULL) { + env_gc_node_t *next = node->next; + enif_free(node); + node = next; + } +} + /* Invariant counters for debugging and leak detection */ py_invariant_counters_t g_counters = {0}; @@ -331,6 +379,9 @@ static void context_destructor(ErlNifEnv *env, void *obj) { /* Close callback pipes if open */ close_pipe_pair(ctx->callback_pipe); + /* Envs released after the interpreter ended: their objects are gone */ + env_gc_discard(ctx); + /* Refcount is zero here, so no interrupt can be in flight. Contexts that * leaked an unresponsive thread keep a reference and never reach this. */ if (ctx->interrupt_mutex_init) { @@ -2127,6 +2178,12 @@ static void ctx_execute_create_local_env(py_context_t *ctx) { /* Copy globals from context to inherit preloaded code */ res->globals = PyDict_Copy(ctx->globals); + if (res->globals != NULL && ctx->uses_own_gil && res->owner_ctx == NULL) { + /* Only this context's thread can release these dicts: the env + * keeps the context alive and hands them back when it dies */ + enif_keep_resource(ctx); + res->owner_ctx = ctx; + } if (res->globals == NULL) { ctx->response_term = enif_make_tuple2(ctx->shared_env, enif_make_atom(ctx->shared_env, "error"), @@ -3071,6 +3128,11 @@ static void *ctx_thread_main_owngil(void *arg) { ctx->event_loop = get_current_interpreter_event_loop(); if (ctx->event_loop != NULL) { enif_keep_resource(ctx->event_loop); + /* Namespaces of processes that die go to this thread for release */ + erlang_event_loop_t *loop = (erlang_event_loop_t *)ctx->event_loop; + pthread_mutex_lock(&loop->namespaces_mutex); + loop->gc_ctx = ctx; + pthread_mutex_unlock(&loop->namespaces_mutex); } /* Create namespace dictionaries */ @@ -3181,6 +3243,7 @@ static void *ctx_thread_main_owngil(void *arg) { * invariant on py_context::interrupt_mutex). */ py_context_exec_enter(ctx); PyEval_RestoreThread(ctx->own_gil_tstate); + env_gc_drain(ctx); ctx_execute_request(ctx); PyEval_SaveThread(); py_context_exec_leave(ctx); @@ -3233,8 +3296,18 @@ static void *ctx_thread_main_owngil(void *arg) { event_loop_detach_interpreter((erlang_event_loop_t *)ctx->event_loop); } + /* No more namespaces handed to us once we stop serving; what was + * handed before is released just below */ + if (ctx->event_loop != NULL) { + erlang_event_loop_t *loop = (erlang_event_loop_t *)ctx->event_loop; + pthread_mutex_lock(&loop->namespaces_mutex); + loop->gc_ctx = NULL; + pthread_mutex_unlock(&loop->namespaces_mutex); + } + /* Cleanup: acquire our OWN_GIL and destroy interpreter */ PyEval_RestoreThread(ctx->own_gil_tstate); + env_gc_drain(ctx); Py_XDECREF(ctx->module_cache); Py_XDECREF(ctx->globals); Py_XDECREF(ctx->locals); @@ -3546,6 +3619,7 @@ static ERL_NIF_TERM nif_context_create(ErlNifEnv *env, int argc, const ERL_NIF_T /* Initialize fields */ ctx->interp_id = atomic_fetch_add(&g_context_id_counter, 1); + atomic_init(&ctx->env_gc_head, NULL); ctx->is_subinterp = use_owngil; atomic_store(&ctx->destroyed, false); atomic_store(&ctx->leaked, false); @@ -4268,6 +4342,7 @@ static ERL_NIF_TERM nif_create_local_env(ErlNifEnv *env, int argc, const ERL_NIF res->globals = NULL; res->locals = NULL; res->interp_id = 0; + res->owner_ctx = NULL; if (!ctx_uses_async_thread(ctx)) { enif_release_resource(res); diff --git a/c_src/py_nif.h b/c_src/py_nif.h index be393c0..c701a97 100644 --- a/c_src/py_nif.h +++ b/c_src/py_nif.h @@ -355,6 +355,17 @@ typedef struct { /* Forward declaration for py_context_t (defined later) */ typedef struct py_context py_context_t; +/** + * @brief Dicts of a dead process-local env, waiting to be released on the + * OWN_GIL context thread that created them (see env_gc_head). + */ +typedef struct env_gc_node { + PyObject *globals; + PyObject *locals; + PyObject *extra; + struct env_gc_node *next; +} env_gc_node_t; + /** * @enum py_request_type_t * @brief Kinds of Python work a context can run @@ -828,6 +839,14 @@ struct py_context { /** @brief Interpreter state for OWN_GIL subinterpreter */ PyInterpreterState *own_gil_interp; + /** @brief Dicts of dead process-local envs of this OWN_GIL context. + * The env destructor runs on any thread and cannot take this + * interpreter's GIL: it pushes here (lock-free), and the context thread + * releases them before its next request and before the interpreter + * ends. Nodes pushed after that are freed by the context destructor, + * their objects having gone with the interpreter. */ + _Atomic(env_gc_node_t *) env_gc_head; + /** @brief Default ErlangEventLoop of the subinterpreter (kept resource, * set by the context thread, read by nif_context_get_event_loop) */ void *event_loop; @@ -1291,8 +1310,18 @@ typedef struct py_env_resource { PyObject *locals; /** @brief Interpreter ID that owns these dicts (0 = main interpreter) */ int64_t interp_id; + /** @brief OWN_GIL context that created the dicts (kept referenced), so + * the destructor can hand them back to its thread; NULL otherwise */ + py_context_t *owner_ctx; } py_env_resource_t; +/** + * @brief Hand objects of an OWN_GIL context's interpreter to its thread for + * release (env_gc_head). Lock-free; callable from any thread without the + * GIL. NULL objects are allowed. + */ +void env_gc_push(py_context_t *ctx, PyObject *globals, PyObject *locals, PyObject *extra); + /** @brief Resource type for py_env_resource_t (process-local Python environment) */ extern ErlNifResourceType *PY_ENV_RESOURCE_TYPE; diff --git a/docs/owngil_internals.md b/docs/owngil_internals.md index c7ae84f..9fcc039 100644 --- a/docs/owngil_internals.md +++ b/docs/owngil_internals.md @@ -290,11 +290,12 @@ while (!shutdown_requested) { ```c py_env_resource_dtor(env, res) { - if (res->pool_slot >= 0) { - // Shared-GIL subinterpreter: DECREF with pool GIL - } else if (res->interp_id != 0) { - // OWN_GIL subinterpreter: skip DECREF - // Py_EndInterpreter cleans up all objects + if (res->owner_ctx != NULL) { + // OWN_GIL env: push the dicts on the context's lock-free list; + // the context thread releases them before its next request + // (env_gc_drain), and before Py_EndInterpreter + env_gc_push(res->owner_ctx, res->globals, res->locals, NULL); + enif_release_resource(res->owner_ctx); } else { // Worker mode: DECREF with main GIL } @@ -556,7 +557,7 @@ pthread_mutex_unlock(&loop->namespaces_mutex); PyGILState_Release(gstate); ``` -For subinterpreters (where `PyGILState_Ensure` cannot be used), cleanup skips `Py_DECREF` - the objects will be freed when the interpreter is destroyed. +For a subinterpreter loop, a dead process's namespace is released by the interpreter that created it: under the main GIL for namespaces made by `event_loop_exec/eval` (they run on the main interpreter), by the owngil context thread (`env_gc_push`) for those made in the subinterpreter. The loop destructor releases the main-interpreter ones; the others go with the interpreter. ### Callback Re-entry Limitation diff --git a/docs/process-bound-envs.md b/docs/process-bound-envs.md index 594a551..248ef06 100644 --- a/docs/process-bound-envs.md +++ b/docs/process-bound-envs.md @@ -451,11 +451,17 @@ This prevents: For the main interpreter (`interp_id == 0`), the destructor acquires the GIL and decrefs the Python dicts normally. -For subinterpreters, the destructor skips `Py_DECREF` because: -1. `PyGILState_Ensure` cannot safely acquire a subinterpreter's GIL -2. The Python objects will be freed when the subinterpreter is destroyed via `Py_EndInterpreter` - -This design prioritizes safety over avoiding minor memory leaks during edge cases. +For an owngil context, the destructor cannot release the dicts itself: it +runs on whatever thread collects the resource, and `PyGILState_Ensure` +cannot take a subinterpreter's GIL. The env keeps its context referenced +and hands the dicts back to it; the context thread releases them before +its next request, and before its interpreter ends. Dicts handed back after +that went with the interpreter and are not touched. + +The per-process namespaces of an event loop (`py_event_loop:exec/eval`) +follow the same rule when their process exits: those created in the main +interpreter are released under the main GIL, those created in an owngil +interpreter are handed to its context thread. ## See Also diff --git a/test/py_context_async_env_SUITE.erl b/test/py_context_async_env_SUITE.erl index 18f13c5..a73289c 100644 --- a/test/py_context_async_env_SUITE.erl +++ b/test/py_context_async_env_SUITE.erl @@ -21,6 +21,10 @@ ]). -export([ + loop_namespace_freed_after_process_exit_owngil/1, + env_freed_after_process_exit_worker/1, + env_freed_after_process_exit_owngil/1, + env_outlives_owngil_context/1, async_env_call_returns_correct_result/1, env_call_does_not_dispatch_timeout/1 ]). @@ -28,7 +32,11 @@ all() -> [ async_env_call_returns_correct_result, - env_call_does_not_dispatch_timeout + env_call_does_not_dispatch_timeout, + env_freed_after_process_exit_worker, + env_freed_after_process_exit_owngil, + env_outlives_owngil_context, + loop_namespace_freed_after_process_exit_owngil ]. init_per_suite(Config) -> @@ -69,3 +77,102 @@ env_call_does_not_dispatch_timeout(_Config) -> true = Elapsed >= 900, true = Elapsed < 5000, ok. + + +%% A process-local env holds its objects until the process exits; then they +%% are released. Each env registers an object in a WeakSet kept on `sys' +%% (shared by all envs of the interpreter): it empties once they are freed. +env_freed_after_process_exit_worker(_Config) -> + envs_freed(worker). + +%% owngil envs used to be kept until the context stopped: the destructor runs +%% on a thread that cannot take that interpreter's GIL. It now hands them to +%% the context thread, which releases them before its next request. +env_freed_after_process_exit_owngil(_Config) -> + case py_nif:owngil_supported() of + true -> envs_freed(owngil); + false -> {skip, "OWN_GIL requires Python 3.14+"} + end. + +envs_freed(Mode) -> + {ok, C} = py_context:new(#{mode => Mode}), + ok = py_context:exec(C, <<"import sys, weakref\nsys._env_probe = weakref.WeakSet()">>), + [begin + {P, M} = spawn_monitor(fun() -> + ok = py:exec(C, <<"import sys\nclass Held: pass\nheld = Held()\nsys._env_probe.add(held)">>), + {ok, true} = py:eval(C, <<"held in sys._env_probe">>) + end), + receive {'DOWN', M, process, P, normal} -> ok end + end || _ <- lists:seq(1, 50)], + ok = wait_probe_empty(C, 100), + py_context:stop(C). + +wait_probe_empty(C, 0) -> + ct:fail({envs_not_freed, py_context:eval(C, <<"len(__import__('sys')._env_probe)">>)}); +wait_probe_empty(C, N) -> + erlang:garbage_collect(), + %% the next request on the context releases what the dead envs held + case py_context:eval(C, <<"len(__import__('sys')._env_probe)">>) of + {ok, 0} -> ok; + _ -> timer:sleep(20), wait_probe_empty(C, N - 1) + end. + +%% A process still holding an owngil env when its context stops: the env is +%% released after the interpreter is gone, without touching it. +env_outlives_owngil_context(_Config) -> + case py_nif:owngil_supported() of + false -> + {skip, "OWN_GIL requires Python 3.14+"}; + true -> + {ok, C} = py_context:new(#{mode => owngil}), + Self = self(), + Holders = [spawn(fun() -> + ok = py:exec(C, <<"blob = 'x' * 100000">>), + Self ! {ready, self()}, + receive go -> ok end + end) || _ <- lists:seq(1, 10)], + [receive {ready, H} -> ok after 5000 -> ct:fail(no_env) end || H <- Holders], + ok = py_context:stop(C), + Mons = [erlang:monitor(process, H) || H <- Holders], + [H ! go || H <- Holders], + [receive {'DOWN', M, process, _, normal} -> ok after 5000 -> ct:fail(holder_stuck) end + || M <- Mons], + erlang:garbage_collect(), + %% the node is fine and new owngil contexts work + {ok, C2} = py_context:new(#{mode => owngil}), + {ok, 2} = py_context:eval(C2, <<"1 + 1">>), + py_context:stop(C2) + end. + + +%% The namespace a process gets on an owngil context's event loop +%% (py_event_loop:exec/eval) is released when the process exits. Those +%% namespaces are created in the main interpreter even on a subinterpreter +%% loop; they used to be dropped without being released. +loop_namespace_freed_after_process_exit_owngil(_Config) -> + case py_nif:owngil_supported() of + false -> + {skip, "OWN_GIL requires Python 3.14+"}; + true -> + {ok, C} = py_context:new(#{mode => owngil}), + {ok, Loop} = py_context:loop_ref(C), + ok = py_event_loop:exec(Loop, <<"import sys, weakref\nsys._loop_probe = weakref.WeakSet()">>), + [begin + {P, M} = spawn_monitor(fun() -> + ok = py_event_loop:exec(Loop, <<"import sys\nclass Held: pass\nheld = Held()\nsys._loop_probe.add(held)">>) + end), + receive {'DOWN', M, process, P, normal} -> ok end + end || _ <- lists:seq(1, 30)], + ok = wait_loop_probe_empty(Loop, 100), + py_context:stop(C) + end. + +wait_loop_probe_empty(Loop, 0) -> + ct:fail({loop_namespaces_not_freed, + py_event_loop:eval(Loop, <<"len(__import__('sys')._loop_probe)">>)}); +wait_loop_probe_empty(Loop, N) -> + erlang:garbage_collect(), + case py_event_loop:eval(Loop, <<"len(__import__('sys')._loop_probe)">>) of + {ok, 0} -> ok; + _ -> timer:sleep(20), wait_loop_probe_empty(Loop, N - 1) + end. From eccda0d3e90cd29be0310ffcf908c9bb7b6a0d65 Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Wed, 30 Sep 2026 19:22:41 +0200 Subject: [PATCH 17/18] Drop a loop's env registration when its process exits A task submitted with the caller's process-local env registers that env on the event loop, and the registration was only dropped with the loop, so every such process left its env behind. The registration now monitors its process and releases the env when it exits; a task holds its own reference while it looks the env up. The env tests define their classes outside the envs, since on free-threaded builds a class created in an env keeps its globals until CPython reclaims it. --- CHANGELOG.md | 5 +++ c_src/py_event_loop.c | 61 ++++++++++++++++++++++++++--- c_src/py_event_loop.h | 4 ++ test/py_context_async_env_SUITE.erl | 55 +++++++++++++++++++++----- 4 files changed, 110 insertions(+), 15 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 379a958..8bb2487 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -40,6 +40,11 @@ and is released before the context's next request. The per-process namespaces of an owngil context's event loop (`py_event_loop:exec/eval`) were not released either; they now are. +- A process that ran an event loop task with its process-local environment + (`py_event_loop:create_task` after `py:exec`, in any mode) left that + environment registered on the loop for the loop's lifetime. The + registration is now dropped, and the environment released, when the + process exits. - On an owngil context, a Python function that called `erlang.call`, where the Erlang callback called the same context again, hung until the request timeout. The context thread waited for the callback on its pipe while the diff --git a/c_src/py_event_loop.c b/c_src/py_event_loop.c index ac4ae95..d089c8e 100644 --- a/c_src/py_event_loop.c +++ b/c_src/py_event_loop.c @@ -699,6 +699,28 @@ void timer_resource_destructor(ErlNifEnv *env, void *obj) { * Per-Process Namespace Management * ============================================================================ */ +/** + * @brief Unlink the pid-to-env mapping of an exited process. + * + * Call with namespaces_mutex held; release the returned env (if any) with + * enif_release_resource after dropping it, since the env destructor may + * take the GIL (GIL before namespaces_mutex). + */ +static void *take_pid_env(erlang_event_loop_t *loop, const ErlNifPid *pid) { + pid_env_mapping_t **pp = &loop->pid_env_head; + while (*pp != NULL) { + if (enif_compare_pids(&(*pp)->pid, pid) == 0) { + pid_env_mapping_t *m = *pp; + void *env_res = m->env; + *pp = m->next; + enif_free(m); + return env_res; + } + pp = &(*pp)->next; + } + return NULL; +} + /** * @brief Down callback for event loop resources (process monitor) * @@ -723,6 +745,7 @@ void event_loop_down(ErlNifEnv *env, void *obj, ErlNifPid *pid, if (!runtime_is_running() || loop->interp_id != 0) { process_namespace_t *main_ns = NULL; pthread_mutex_lock(&loop->namespaces_mutex); + void *dead_env = take_pid_env(loop, pid); process_namespace_t **pp = &loop->namespaces_head; while (*pp != NULL) { @@ -756,6 +779,9 @@ void event_loop_down(ErlNifEnv *env, void *obj, ErlNifPid *pid, PyGILState_Release(gstate); enif_free(main_ns); } + if (dead_env != NULL) { + enif_release_resource(dead_env); + } return; } @@ -767,6 +793,8 @@ void event_loop_down(ErlNifEnv *env, void *obj, ErlNifPid *pid, PyGILState_STATE gstate = PyGILState_Ensure(); pthread_mutex_lock(&loop->namespaces_mutex); + void *dead_env = take_pid_env(loop, pid); + /* Find and remove namespace for this pid */ process_namespace_t **pp = &loop->namespaces_head; while (*pp != NULL) { @@ -785,6 +813,10 @@ void event_loop_down(ErlNifEnv *env, void *obj, ErlNifPid *pid, } pthread_mutex_unlock(&loop->namespaces_mutex); + if (dead_env != NULL) { + /* The env destructor takes the GIL again: re-entrant here */ + enif_release_resource(dead_env); + } PyGILState_Release(gstate); } @@ -2670,8 +2702,8 @@ ERL_NIF_TERM nif_submit_task(ErlNifEnv *env, int argc, * @param env_res Environment resource (will be kept via enif_keep_resource) * @return true on success, false on allocation failure */ -static bool register_pid_env(erlang_event_loop_t *loop, const ErlNifPid *pid, - void *env_res) { +static bool register_pid_env(ErlNifEnv *caller_env, erlang_event_loop_t *loop, + const ErlNifPid *pid, void *env_res) { pthread_mutex_lock(&loop->namespaces_mutex); /* Check if mapping already exists */ @@ -2696,6 +2728,14 @@ static bool register_pid_env(erlang_event_loop_t *loop, const ErlNifPid *pid, mapping->pid = *pid; mapping->env = env_res; mapping->refcount = 1; + + /* Dropped with its env when the process exits (event_loop_down). A + * process already gone gets no mapping. */ + if (enif_monitor_process(caller_env, loop, pid, &mapping->monitor) != 0) { + enif_free(mapping); + pthread_mutex_unlock(&loop->namespaces_mutex); + return false; + } mapping->next = loop->pid_env_head; loop->pid_env_head = mapping; @@ -2709,6 +2749,9 @@ static bool register_pid_env(erlang_event_loop_t *loop, const ErlNifPid *pid, /** * @brief Look up env for a PID * + * The env is returned with a reference of its own (the process may exit + * and drop the mapping meanwhile): release it with enif_release_resource. + * * @param loop Event loop containing the mapping registry * @param pid PID to look up * @return Environment resource or NULL if not found @@ -2720,6 +2763,7 @@ static void *lookup_pid_env(erlang_event_loop_t *loop, const ErlNifPid *pid) { while (mapping != NULL) { if (enif_compare_pids(&mapping->pid, pid) == 0) { void *env_res = mapping->env; + enif_keep_resource(env_res); pthread_mutex_unlock(&loop->namespaces_mutex); return env_res; } @@ -2767,7 +2811,7 @@ ERL_NIF_TERM nif_submit_task_with_env(ErlNifEnv *env, int argc, } /* Register the env for this PID (increments refcount if exists) */ - if (!register_pid_env(loop, &caller_pid, env_res)) { + if (!register_pid_env(env, loop, &caller_pid, env_res)) { return make_error(env, "env_registration_failed"); } @@ -3343,9 +3387,6 @@ ERL_NIF_TERM nif_process_ready_tasks(ErlNifEnv *env, int argc, continue; } - /* Look up env by PID (registered via submit_task_with_env) */ - py_env_resource_t *task_env = (py_env_resource_t *)lookup_pid_env(loop, &caller_pid); - /* Convert module/func to C strings (stack buffers to avoid alloc overhead) */ char module_name[CALLABLE_NAME_MAX]; char func_name[CALLABLE_NAME_MAX]; @@ -3364,6 +3405,9 @@ ERL_NIF_TERM nif_process_ready_tasks(ErlNifEnv *env, int argc, /* Look up function - check task_env first, then process namespace, then import */ PyObject *func = NULL; + /* Env registered via submit_task_with_env, held for this lookup only */ + py_env_resource_t *task_env = (py_env_resource_t *)lookup_pid_env(loop, &caller_pid); + /* First, check the passed env's globals (from py:exec) */ if (task_env != NULL && task_env->globals != NULL) { if (strcmp(module_name, "__main__") == 0 || @@ -3375,6 +3419,11 @@ ERL_NIF_TERM nif_process_ready_tasks(ErlNifEnv *env, int argc, } } + if (task_env != NULL) { + enif_release_resource(task_env); + task_env = NULL; + } + /* Fallback to process namespace and cache/import */ if (func == NULL) { func = get_function_for_task(loop, ns, module_name, func_name); diff --git a/c_src/py_event_loop.h b/c_src/py_event_loop.h index fce2fb1..db7220a 100644 --- a/c_src/py_event_loop.h +++ b/c_src/py_event_loop.h @@ -150,6 +150,10 @@ typedef struct pid_env_mapping { /** @brief Reference count for this mapping (multiple tasks may use it) */ int refcount; + /** @brief Monitor on the owning process: the mapping and its env are + * dropped when it exits (event_loop_down) */ + ErlNifMonitor monitor; + /** @brief Next mapping in linked list */ struct pid_env_mapping *next; } pid_env_mapping_t; diff --git a/test/py_context_async_env_SUITE.erl b/test/py_context_async_env_SUITE.erl index a73289c..0562949 100644 --- a/test/py_context_async_env_SUITE.erl +++ b/test/py_context_async_env_SUITE.erl @@ -22,6 +22,7 @@ -export([ loop_namespace_freed_after_process_exit_owngil/1, + loop_task_env_freed_after_process_exit/1, env_freed_after_process_exit_worker/1, env_freed_after_process_exit_owngil/1, env_outlives_owngil_context/1, @@ -36,7 +37,8 @@ all() -> env_freed_after_process_exit_worker, env_freed_after_process_exit_owngil, env_outlives_owngil_context, - loop_namespace_freed_after_process_exit_owngil + loop_namespace_freed_after_process_exit_owngil, + loop_task_env_freed_after_process_exit ]. init_per_suite(Config) -> @@ -82,6 +84,9 @@ env_call_does_not_dispatch_timeout(_Config) -> %% A process-local env holds its objects until the process exits; then they %% are released. Each env registers an object in a WeakSet kept on `sys' %% (shared by all envs of the interpreter): it empties once they are freed. +%% The class is defined in the context, not in the envs: on free-threaded +%% builds a class created in an env keeps its globals until CPython reclaims +%% the class on its own schedule, which is not what is tested here. env_freed_after_process_exit_worker(_Config) -> envs_freed(worker). @@ -96,15 +101,16 @@ env_freed_after_process_exit_owngil(_Config) -> envs_freed(Mode) -> {ok, C} = py_context:new(#{mode => Mode}), - ok = py_context:exec(C, <<"import sys, weakref\nsys._env_probe = weakref.WeakSet()">>), + ok = py_context:exec(C, <<"import sys, weakref\nsys._env_probe = weakref.WeakSet()\n" + "class Held: pass\nsys._EnvHeld = Held">>), [begin {P, M} = spawn_monitor(fun() -> - ok = py:exec(C, <<"import sys\nclass Held: pass\nheld = Held()\nsys._env_probe.add(held)">>), + ok = py:exec(C, <<"import sys\nheld = sys._EnvHeld()\nsys._env_probe.add(held)">>), {ok, true} = py:eval(C, <<"held in sys._env_probe">>) end), receive {'DOWN', M, process, P, normal} -> ok end end || _ <- lists:seq(1, 50)], - ok = wait_probe_empty(C, 100), + ok = wait_probe_empty(C, 250), py_context:stop(C). wait_probe_empty(C, 0) -> @@ -112,7 +118,7 @@ wait_probe_empty(C, 0) -> wait_probe_empty(C, N) -> erlang:garbage_collect(), %% the next request on the context releases what the dead envs held - case py_context:eval(C, <<"len(__import__('sys')._env_probe)">>) of + case py_context:eval(C, <<"(__import__('gc').collect(), len(__import__('sys')._env_probe))[1]">>) of {ok, 0} -> ok; _ -> timer:sleep(20), wait_probe_empty(C, N - 1) end. @@ -156,14 +162,15 @@ loop_namespace_freed_after_process_exit_owngil(_Config) -> true -> {ok, C} = py_context:new(#{mode => owngil}), {ok, Loop} = py_context:loop_ref(C), - ok = py_event_loop:exec(Loop, <<"import sys, weakref\nsys._loop_probe = weakref.WeakSet()">>), + ok = py_event_loop:exec(Loop, <<"import sys, weakref\nsys._loop_probe = weakref.WeakSet()\n" + "class Held: pass\nsys._LoopHeld = Held">>), [begin {P, M} = spawn_monitor(fun() -> - ok = py_event_loop:exec(Loop, <<"import sys\nclass Held: pass\nheld = Held()\nsys._loop_probe.add(held)">>) + ok = py_event_loop:exec(Loop, <<"import sys\nheld = sys._LoopHeld()\nsys._loop_probe.add(held)">>) end), receive {'DOWN', M, process, P, normal} -> ok end end || _ <- lists:seq(1, 30)], - ok = wait_loop_probe_empty(Loop, 100), + ok = wait_loop_probe_empty(Loop, 250), py_context:stop(C) end. @@ -172,7 +179,37 @@ wait_loop_probe_empty(Loop, 0) -> py_event_loop:eval(Loop, <<"len(__import__('sys')._loop_probe)">>)}); wait_loop_probe_empty(Loop, N) -> erlang:garbage_collect(), - case py_event_loop:eval(Loop, <<"len(__import__('sys')._loop_probe)">>) of + case py_event_loop:eval(Loop, <<"(__import__('gc').collect(), len(__import__('sys')._loop_probe))[1]">>) of {ok, 0} -> ok; _ -> timer:sleep(20), wait_loop_probe_empty(Loop, N - 1) end. + + +%% A task submitted to the event loop from a process that has a local env +%% (py_event_loop:create_task after py:exec) registers that env for the +%% process. The registration used to keep the env for the loop's lifetime; +%% it is dropped when the process exits. +loop_task_env_freed_after_process_exit(_Config) -> + ok = py:exec(<<"import sys, weakref\nsys._task_env_probe = weakref.WeakSet()\n" + "class Held: pass\nsys._TaskHeld = Held">>), + [begin + {P, M} = spawn_monitor(fun() -> + ok = py:exec(<<"import sys\nheld = sys._TaskHeld()\nsys._task_env_probe.add(held)">>), + %% the process has an env, so the task registers it; the task + %% itself calls a library coroutine (no function defined in the + %% env, see envs_freed/1 about free-threaded builds) + Ref = py_event_loop:create_task(asyncio, sleep, [0]), + {ok, none} = py_event_loop:await(Ref, 5000) + end), + receive {'DOWN', M, process, P, normal} -> ok after 10000 -> ct:fail(task_hung) end + end || _ <- lists:seq(1, 20)], + ok = wait_task_probe_empty(250). + +wait_task_probe_empty(0) -> + ct:fail({task_envs_not_freed, py:eval(<<"len(__import__('sys')._task_env_probe)">>)}); +wait_task_probe_empty(N) -> + erlang:garbage_collect(), + case py:eval(<<"(__import__('gc').collect(), len(__import__('sys')._task_env_probe))[1]">>) of + {ok, 0} -> ok; + _ -> timer:sleep(20), wait_task_probe_empty(N - 1) + end. From 76ea042f97f9ab48d721a6a3229380ba7ab6726a Mon Sep 17 00:00:00 2001 From: Benoit Chesneau Date: Wed, 30 Sep 2026 19:38:39 +0200 Subject: [PATCH 18/18] Collect every process before counting released envs An env reference sent to a context process stays on its heap until it collects and keeps the env alive until then, so the env tests counted envs an idle context had not yet let go of. They now collect every process first; the leaks they guard against are held from C and still fail them. --- test/py_context_async_env_SUITE.erl | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/test/py_context_async_env_SUITE.erl b/test/py_context_async_env_SUITE.erl index 0562949..893739a 100644 --- a/test/py_context_async_env_SUITE.erl +++ b/test/py_context_async_env_SUITE.erl @@ -116,7 +116,7 @@ envs_freed(Mode) -> wait_probe_empty(C, 0) -> ct:fail({envs_not_freed, py_context:eval(C, <<"len(__import__('sys')._env_probe)">>)}); wait_probe_empty(C, N) -> - erlang:garbage_collect(), + gc_all(), %% the next request on the context releases what the dead envs held case py_context:eval(C, <<"(__import__('gc').collect(), len(__import__('sys')._env_probe))[1]">>) of {ok, 0} -> ok; @@ -143,7 +143,7 @@ env_outlives_owngil_context(_Config) -> [H ! go || H <- Holders], [receive {'DOWN', M, process, _, normal} -> ok after 5000 -> ct:fail(holder_stuck) end || M <- Mons], - erlang:garbage_collect(), + gc_all(), %% the node is fine and new owngil contexts work {ok, C2} = py_context:new(#{mode => owngil}), {ok, 2} = py_context:eval(C2, <<"1 + 1">>), @@ -178,7 +178,7 @@ wait_loop_probe_empty(Loop, 0) -> ct:fail({loop_namespaces_not_freed, py_event_loop:eval(Loop, <<"len(__import__('sys')._loop_probe)">>)}); wait_loop_probe_empty(Loop, N) -> - erlang:garbage_collect(), + gc_all(), case py_event_loop:eval(Loop, <<"(__import__('gc').collect(), len(__import__('sys')._loop_probe))[1]">>) of {ok, 0} -> ok; _ -> timer:sleep(20), wait_loop_probe_empty(Loop, N - 1) @@ -208,8 +208,15 @@ loop_task_env_freed_after_process_exit(_Config) -> wait_task_probe_empty(0) -> ct:fail({task_envs_not_freed, py:eval(<<"len(__import__('sys')._task_env_probe)">>)}); wait_task_probe_empty(N) -> - erlang:garbage_collect(), + gc_all(), case py:eval(<<"(__import__('gc').collect(), len(__import__('sys')._task_env_probe))[1]">>) of {ok, 0} -> ok; _ -> timer:sleep(20), wait_task_probe_empty(N - 1) end. + +%% An env reference sent to a context process (py_context:exec(Ctx, Code, +%% EnvRef)) stays on that process's heap until it collects, and keeps the env +%% alive until then: collect every process before counting. +gc_all() -> + [erlang:garbage_collect(P) || P <- erlang:processes()], + ok.