Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion Project.toml
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
name = "ReTestItems"
uuid = "817f1d60-ba6b-4fd5-9520-3cf149f6a823"
version = "1.35.2"
version = "1.36.0"

[deps]
Dates = "ade2ca70-3891-5945-98fb-dc099432e06a"
Expand Down
24 changes: 24 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -229,6 +229,30 @@ end

If a test-items set the `failfast` then that value takes precedence over the `testitem_failfast` keyword passed to `runtests`.

#### Running expensive test-items first

If a test-item takes much longer to run than the others, it can say so using the `cost` keyword.
Test-items that declare a cost are run before those that don't, most expensive first.

```julia
@testitem "slow integration test" cost=450 begin
@test long_running_thing()
end
```

Starting the most expensive test-items first shortens the whole test run, since a long test-item that starts near the end of the run leaves every other worker idle waiting for it to finish.
A cost is a number of nominal seconds. Only the relative size of costs matters, so costs need not be accurate — a rough measurement is enough to get most of the benefit — but they should be on a consistent scale within a project.

A cost can also be computed from the configuration of the test run, by passing a function.
It is called once per test-item, before any test-item runs, with a `NamedTuple` holding `nworkers::Int`, the number of worker processes (`0` when tests run serially), and `nworker_threads::Int`, the number of threads each test-item will run with.
For example, a test-item with 15 seconds of fixed cost plus 450 seconds of work that its worker can parallelize across threads:

```julia
@testitem "slow integration test" cost=(cfg -> 15 + 450 / cfg.nworker_threads) begin
@test long_running_thing()
end
```

#### Post-testitem hook

If there is something that should be checked after every single `@testitem`, then it's possible to pass an expression to `runtests` using the `test_end_expr` keyword.
Expand Down
93 changes: 84 additions & 9 deletions src/ReTestItems.jl
Original file line number Diff line number Diff line change
Expand Up @@ -262,6 +262,8 @@ will be run.
- `failures_first::Bool=true`: if `true`, first runs test items that failed the last time
they ran, followed by new test items, followed by test items that passed the last time they ran.
Can also be set using the `RETESTITEMS_FAILURES_FIRST` environment variable.
Within each of those groups, test items that declare a `cost` run before those that
don't, most expensive first; see the `cost` keyword of `@testitem`.
"""
function runtests end

Expand Down Expand Up @@ -399,6 +401,87 @@ function _runtests(ti_filter, paths, cfg::_Config)
end
end

# Stands in for the cost of a test item that declares none. Lower than every valid cost, so
# that once costs are negated to sort descending, those test items sort last.
const _NO_COST = -Inf

# The run configuration passed to a `@testitem`'s `cost` function. `nworker_threads` is
# the number of (default-threadpool) threads each test item will run with, so a cost can
# scale with the parallelism available to the test item: the validated "N" or "N,M"
# setting when there are workers, and this process's thread count when running serially
# (the setting is unused then).
function _cost_config(cfg::_Config)
nthreads = if cfg.nworkers == 0
Threads.nthreads()
else
parse(Int, first(split(cfg.nworker_threads, ',')))
end
return (; nworkers=cfg.nworkers, nworker_threads=nthreads)
end

_scheduling_cost(ti::TestItem) = (c = ti.cost[]; c isa Real ? Float64(c) : _NO_COST)

# The one-argument form is the documented interface; a function of no arguments is accepted
# too, for a cost that doesn't depend on the run configuration. If neither form is
# applicable (e.g. the one-argument method demands some other type), still call the
# one-argument form, so the resulting `MethodError` points at the documented interface.
function _call_cost(f::Function, cost_cfg::NamedTuple)
applicable(f, cost_cfg) && return f(cost_cfg)
applicable(f) && return f()
return f(cost_cfg)
end

function _call_cost_function(f::Function, ti::TestItem, cost_cfg::NamedTuple)
# The test files were included by this same `runtests` call, so a cost function defined
# in a test file is newer than the world age we are running in, and both the call and
# the check of which form it takes have to happen in the latest world.
try
return _validated_cost_result(Base.invokelatest(_call_cost, f, cost_cfg))
catch
@error "Error evaluating `cost` for test item $(repr(ti.name)) at $(ti.file):$(ti.line)"
rethrow()
end
end

# Replace each `Function` cost by the number it returns. Costs are resolved here, once per
# run and before any test item is sent to a worker, so that a cost function is called
# exactly once and never leaves the coordinator process.
# Returns whether any test item declares a cost.
function _resolve_costs!(testitems::Vector{TestItem}, cfg::_Config)
any_cost = false
cost_cfg = _cost_config(cfg)
for ti in testitems
cost = ti.cost[]
if cost isa Function
cost = _call_cost_function(cost, ti, cost_cfg)
ti.cost[] = cost
end
any_cost |= !isnothing(cost)
end
return any_cost
end

# Put the queue in the order we want test items picked up in: test items that failed the
# last time they ran first (`failures_first`), then most expensive first.
# Returns whether the queue is now sorted, i.e. whether workers should start from the front
# of the queue rather than from evenly spaced positions.
function _sort_testitems!(testitems::TestItems, cfg::_Config)
any_cost = _resolve_costs!(testitems.testitems, cfg)
by_status = cfg.failures_first && !isempty(GLOBAL_TEST_STATUS)
(any_cost || by_status) || return false
# `number` is unique, so the key is a total order and the resulting order is
# deterministic whether or not the sort algorithm is stable.
sort!(testitems.testitems; by=ti -> (
by_status ? _status_when_last_seen(ti) : _UNSEEN,
-_scheduling_cost(ti),
ti.number[],
))
foreach(enumerate(testitems.testitems)) do (i, ti)
ti.number[] = i # reset number to match new order
end
return true
end

function _runtests_in_current_env(
ti_filter, paths, projectfile::String, cfg::_Config
)
Expand All @@ -419,15 +502,7 @@ function _runtests_in_current_env(
@info "Scheduling $ntestitems tests on pid $(Libc.getpid())" *
(nworkers == 0 ? "" : " with $nworkers worker processes and $nworker_threads threads per worker.")
try
if cfg.failures_first && !isempty(GLOBAL_TEST_STATUS)
sort!(testitems.testitems; by=_status_when_last_seen)
foreach(enumerate(testitems.testitems)) do (i, ti)
ti.number[] = i # reset number to match new order
end
is_sorted_queue = true
else
is_sorted_queue = false
end
is_sorted_queue = _sort_testitems!(testitems, cfg)
if nworkers == 0
length(cfg.worker_init_expr.args) > 0 && error("worker_init_expr is set, but will not run because number of workers is 0.")
# This is where we disable printing for the serial executor case.
Expand Down
75 changes: 71 additions & 4 deletions src/macros.jl
Original file line number Diff line number Diff line change
Expand Up @@ -122,6 +122,11 @@ struct TestItem
timeout::Union{Int,Nothing} # in seconds
skip::Union{Bool,Expr}
failfast::Union{Bool,Nothing}
# Nominal seconds this test item takes to run, used to schedule expensive test items
# first; `nothing` if the test item declares no cost. A `Function` cost is called by
# the runtests coordinator and replaced by the number it returns, before any test item
# is sent to a worker, so a worker only ever sees a number or `nothing`.
cost::Base.RefValue{Union{Nothing,Float64,Function}}
file::String
line::Int
project_root::String
Expand All @@ -134,10 +139,33 @@ struct TestItem
stats::Vector{PerfStats} # populated when the test item is finished running
scheduled_for_evaluation::ScheduledForEvaluation # to keep track of whether the test item has been scheduled for evaluation
end
function TestItem(number, name, id, tags, default_imports, setups, retries, timeout, skip, failfast, file, line, project_root, code)
# Normalize what `@testitem` was given as a `cost` to `nothing`, a `Float64`, or the
# `Function` that will compute it.
_invalid_cost(x) = throw(ArgumentError("`cost` must be a `Real` or a `Function`, got `cost=$(repr(x))`"))
_validated_cost(::Nothing) = nothing
_validated_cost(f::Function) = f
function _validated_cost(x::Real)
(isfinite(x) && x >= 0) || throw(ArgumentError("`cost` must be a finite, non-negative number, got `cost=$x`"))
return Float64(x)
end
_validated_cost(x::Bool) = _invalid_cost(x) # `Bool <: Real`, but a cost of `true` is a mistake
_validated_cost(x) = _invalid_cost(x)

# Validate what a `Function` cost returned. Unlike `_validated_cost`, a `Function` is not
# accepted: resolving a cost must produce the number (or `nothing`) that the scheduler
# and the workers will see.
_invalid_cost_result(x) = throw(ArgumentError("`cost` function must return a `Real` or `nothing`, got `$(repr(x))`"))
_validated_cost_result(::Nothing) = nothing
_validated_cost_result(x::Real) = _validated_cost(x)
_validated_cost_result(x::Bool) = _invalid_cost_result(x)
_validated_cost_result(x) = _invalid_cost_result(x)

function TestItem(number, name, id, tags, default_imports, setups, retries, timeout, skip, failfast, cost, file, line, project_root, code)
_id = @something(id, repr(hash(name, hash(relpath(file, project_root)))))
return TestItem(
number, name, _id, tags, default_imports, setups, retries, timeout, skip, failfast, file, line, project_root, code,
number, name, _id, tags, default_imports, setups, retries, timeout, skip, failfast,
Ref{Union{Nothing,Float64,Function}}(_validated_cost(cost)),
file, line, project_root, code,
TestSetup[],
Ref{Int}(0),
DefaultTestSet[],
Expand All @@ -149,7 +177,7 @@ function TestItem(number, name, id, tags, default_imports, setups, retries, time
end

"""
@testitem "name" [tags=[] setup=[] retries=0 skip=false default_imports=true] begin
@testitem "name" [tags=[] setup=[] retries=0 skip=false cost=nothing default_imports=true] begin
# code that will be run as tests
end

Expand Down Expand Up @@ -252,6 +280,38 @@ If a `@testitem` should stop running on the first test failure, then you can set
@test true
@test error("oops")
end

If a `@testitem` takes much longer to run than the others, it can declare how long by
passing the `cost` keyword. Test items that declare a cost are run before those that don't,
most expensive first. Starting the most expensive test items first shortens the whole test
run, since a long test item that starts near the end of the run leaves every other worker
idle waiting for it.

@testitem "slow integration test" cost=450 begin
@test long_running_thing()
end

A cost is a number of nominal seconds. Only the relative size of costs matters, so costs
need only be on a consistent scale within a project; costs need not be accurate, and a
rough measurement is enough to get most of the benefit. Test items that declare no cost run
after all test items that do, in the order they would otherwise have run in.

A cost can also be computed from the configuration of the test run, by passing a function.
For example, a test item with 15 seconds of fixed cost plus 450 seconds of work that its
worker can parallelize across threads:

@testitem "slow integration test" cost=(cfg -> 15 + 450 / cfg.nworker_threads) begin
@test long_running_thing()
end

The function is passed a `NamedTuple` with fields `nworkers::Int`, the number of worker
processes (`0` when tests run serially in the coordinator process), and
`nworker_threads::Int`, the number of threads each test item will run with (from the
`nworker_threads` setting, or this process's thread count when running serially). The
function is called exactly once per test item, in the coordinator process, before any
test item starts running; it is never called in a worker process, so it cannot measure a
worker's environment. A function taking no arguments is also accepted, as is returning
`nothing` to declare no cost after all.
"""
macro testitem(nm, exs...)
default_imports = true
Expand All @@ -261,6 +321,7 @@ macro testitem(nm, exs...)
setup = Any[]
skip = false
failfast = nothing
cost = nothing
_id = nothing
_run = true # useful for testing `@testitem` itself
_source = QuoteNode(__source__)
Expand Down Expand Up @@ -299,6 +360,12 @@ macro testitem(nm, exs...)
elseif kw == :failfast
failfast = ex.args[2]
@assert failfast isa Bool "`failfast` keyword must be passed a `Bool`. Got `failfast=$failfast`"
elseif kw == :cost
cost = ex.args[2]
# A `Function` is written as an expression, so anything but a number is
# only checked here for being expression-shaped; the value it evaluates to
# is validated when the test item is created.
@assert cost isa Union{Real,Symbol,Expr} "`cost` keyword must be passed a `Real` or a `Function`. Got `cost=$cost`"
elseif kw == :_id
_id = ex.args[2]
# This will always be written to the JUnit XML as a String, require the user
Expand All @@ -324,7 +391,7 @@ macro testitem(nm, exs...)
ti = gensym(:ti)
esc(quote
let $ti = $TestItem(
$Ref(0), $nm, $_id, $tags, $default_imports, $setup, $retries, $timeout, $skip, $failfast,
$Ref(0), $nm, $_id, $tags, $default_imports, $setup, $retries, $timeout, $skip, $failfast, $cost,
$String($_source.file), $_source.line,
$gettls(:__RE_TEST_PROJECT__, "."),
$q,
Expand Down
46 changes: 39 additions & 7 deletions test/integrationtests.jl
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,15 @@ const TEST_PKGS = ("NoDeps.jl", "TestsInSrc.jl", "TestProjectFile.jl", "TestEndE

include(joinpath(_TEST_DIR, "_integration_test_tools.jl"))

# The order in which a `runtests` call ran its test items, reconstructed from the
# "START (i/n)" messages in the captured logs.
function testitems_runorder(logstr::String)
re = r"START \(\s*(?<num>\d+)/\d+\) test item \"(?<name>.*)\""
names = [String(m[:name]) for m in eachmatch(re, logstr)]
order = [parse(Int, m[:num]) for m in eachmatch(re, logstr)]
return names[order]
end

# Run `f` in the given package's environment and inside a `testset` which doesn't let
# the package's test failures/errors cause ReTestItems' tests to fail/error.
function with_test_package(f, name)
Expand Down Expand Up @@ -1518,13 +1527,6 @@ end

@testset "failures_first" verbose=true begin
using IOCapture
# we use logs to tell us the order in which tests were run.
function testitems_runorder(logstr::String)
re = r"START \((?<num>\d)/\d\) test item \"(?<name>.*)\""
names = [String(m[:name]) for m in eachmatch(re, logstr)]
order = [parse(Int, m[:num]) for m in eachmatch(re, logstr)]
return names[order]
end
file = joinpath(TEST_FILES_DIR, "_failures_first_tests.jl")
@testset for nworkers in (0, 1)
ReTestItems.reset_test_status!()
Expand Down Expand Up @@ -1588,6 +1590,36 @@ end
end
end

@testset "testitem cost" verbose=true begin
using IOCapture
file = joinpath(TEST_FILES_DIR, "_cost_tests.jl")
# Most expensive first, then the test items declaring no cost, in the order they
# appear in the file.
expected = ["b. cost 100", "e. cost function", "d. cost 50", "a. no cost", "c. no cost"]
@testset for nworkers in (0, 1)
ReTestItems.reset_test_status!()
c = IOCapture.capture() do
encased_testset(()->runtests(file; nworkers))
end
results = c.value
@test n_tests(results) == 5
@test n_passed(results) == 5
@test testitems_runorder(c.output) == expected
end
@testset "workers start on the most expensive test items" begin
ReTestItems.reset_test_status!()
c = IOCapture.capture() do
encased_testset(()->runtests(file; nworkers=2))
end
results = c.value
@test n_tests(results) == 5
@test n_passed(results) == 5
tis = testitems_runorder(c.output)
@test Set(tis[1:2]) == Set(["b. cost 100", "e. cost function"])
end
ReTestItems.reset_test_status!()
end

# https://github.com/JuliaTesting/ReTestItems.jl/issues/228
@testset "issues/228 workers always activate test env" begin
using ReTestItems
Expand Down
Loading
Loading