daemon: sched/ — the concurrency governor + schedule-window evaluation (build step 4)
The scheduling brain, built against CORE's headers (vdm/engine.hpp,
vdm/segment/budget.hpp) — signatures only; the Engine bodies land in
CORE stage 8 and the Scheduler that wires governor <-> store <-> engine
<-> timer comes after that (daemon/docs/deferrals.md D4).
- sched/governor — a pure decision function. In: a snapshot of every
task's coarse RunState and every queue's state (schedule windows
pre-resolved). Out: {to_start, to_resume, to_pause, pause_reasons,
priority_order}. Enforces, all in TASK units per ADR 0011 §1:
connection.maxConcurrentDownloads; the min(that, maxActiveSegments)
clamp (§2); Queue.maxConcurrent; the per-host task cap (§4); and a
stopped queue / closed window runs nothing. Never touches a task
paused for `user` or CORE's `auto` (ADR 0013 §3) — only Schedule /
QueueStopped / AdmissionReconcile are auto-resumable. Deterministic:
main-list before queued-in-queue, then queue order, then FIFO, then
task_id.
- sched/schedule_window — window_open(Schedule, local tm): disabled =>
always open; `once` => date + time match; `periodic` => weekday in
daysOfWeek (empty = every day) + time in [start, stop); null start =>
midnight, null stop => end of day, stop < start => overnight window.
Pure; re-evaluated every tick, no cached instants.
- veloxd_sched static lib; veloxd links it (nothing calls it yet).
Tests (ASan+UBSan and TSan clean): veloxd.sched_window (10 window
cases incl. overnight, once, null bounds), veloxd.sched_governor
(global/clamp/per-queue/per-host caps, stop vs window pause reasons,
resume-only-governor-reasons, auth-pause untouched, admission
reconcile, determinism under shuffled input). 32 daemon/cli tests
green; full tree green.
Co-Authored-By: Claude Sonnet 5 <[email protected]>
Claude-Session: https://claude.ai/code/session_01Upd9WhG9oppieig5nRDLig
This commit is contained in:
@@ -0,0 +1,158 @@
|
||||
#include "sched/governor.hpp"
|
||||
|
||||
#include <algorithm>
|
||||
#include <limits>
|
||||
#include <tuple>
|
||||
#include <unordered_map>
|
||||
|
||||
namespace velox::daemon::sched {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr std::int64_t kUnlimited = std::numeric_limits<std::int64_t>::max();
|
||||
|
||||
// A total order over tasks: main-list (no queue) outranks queued-in-queue; within a queue
|
||||
// by run order; then FIFO by admission rank; task_id breaks any remaining tie so the
|
||||
// result is deterministic. "Lower" == higher priority == admitted first, paused last.
|
||||
auto priority_key(const TaskView& t) {
|
||||
const int tier = t.queue_id ? 1 : 0;
|
||||
const std::string& q = t.queue_id ? *t.queue_id : "";
|
||||
return std::make_tuple(tier, q, t.queue_position, t.admit_rank, t.task_id);
|
||||
}
|
||||
|
||||
bool less_priority(const TaskView* a, const TaskView* b) {
|
||||
return priority_key(*a) < priority_key(*b);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
Decision Governor::evaluate(const std::vector<TaskView>& tasks,
|
||||
const std::vector<QueueView>& queues) const {
|
||||
Decision d;
|
||||
|
||||
std::unordered_map<std::string, const QueueView*> qby;
|
||||
for (const auto& q : queues) qby.emplace(q.queue_id, &q);
|
||||
|
||||
const std::int64_t global_cap =
|
||||
std::min(cfg_.max_concurrent_downloads, cfg_.max_active_segments);
|
||||
|
||||
auto queue_runnable = [&](const std::optional<std::string>& qid) -> bool {
|
||||
if (!qid) return true; // the main list is always "runnable"
|
||||
const auto it = qby.find(*qid);
|
||||
return it != qby.end() && it->second->running && it->second->window_open;
|
||||
};
|
||||
auto queue_cap = [&](const std::string& qid) -> std::int64_t {
|
||||
const auto it = qby.find(qid);
|
||||
return it == qby.end() ? 0 : std::max<std::int64_t>(1, it->second->max_concurrent);
|
||||
};
|
||||
auto host_cap = [&](const std::string& host) -> std::int64_t {
|
||||
if (host.empty()) return kUnlimited;
|
||||
const auto it = cfg_.host_caps.find(host);
|
||||
return it == cfg_.host_caps.end() ? kUnlimited : it->second;
|
||||
};
|
||||
|
||||
// ---- phase 1: pause running tasks whose queue can no longer host them -------------
|
||||
std::vector<const TaskView*> running;
|
||||
for (const auto& t : tasks) {
|
||||
if (t.run_state == RunState::Running) running.push_back(&t);
|
||||
}
|
||||
std::sort(running.begin(), running.end(), less_priority);
|
||||
|
||||
std::vector<const TaskView*> survivors;
|
||||
for (const auto* t : running) {
|
||||
if (t->queue_id && !queue_runnable(t->queue_id)) {
|
||||
const auto it = qby.find(*t->queue_id);
|
||||
const bool stopped_not_windowed =
|
||||
it != qby.end() && it->second->running && !it->second->window_open;
|
||||
const PauseReason why =
|
||||
stopped_not_windowed ? PauseReason::Schedule : PauseReason::QueueStopped;
|
||||
d.to_pause.push_back(t->task_id);
|
||||
d.pause_reasons.emplace(t->task_id, why);
|
||||
} else {
|
||||
survivors.push_back(t);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- phase 2a: per-queue reconcile (a lowered Queue.maxConcurrent) ---------------
|
||||
{
|
||||
std::unordered_map<std::string, std::int64_t> per_queue;
|
||||
std::vector<const TaskView*> kept;
|
||||
for (const auto* t : survivors) { // already priority-sorted
|
||||
if (!t->queue_id) {
|
||||
kept.push_back(t);
|
||||
continue;
|
||||
}
|
||||
auto& n = per_queue[*t->queue_id];
|
||||
if (n < queue_cap(*t->queue_id)) {
|
||||
++n;
|
||||
kept.push_back(t);
|
||||
} else {
|
||||
d.to_pause.push_back(t->task_id);
|
||||
d.pause_reasons.emplace(t->task_id, PauseReason::AdmissionReconcile);
|
||||
}
|
||||
}
|
||||
survivors.swap(kept);
|
||||
}
|
||||
|
||||
// ---- phase 2b: global reconcile (a lowered maxConcurrentDownloads / clamp) -------
|
||||
if (static_cast<std::int64_t>(survivors.size()) > global_cap) {
|
||||
for (std::size_t i = static_cast<std::size_t>(std::max<std::int64_t>(global_cap, 0));
|
||||
i < survivors.size(); ++i) {
|
||||
d.to_pause.push_back(survivors[i]->task_id);
|
||||
d.pause_reasons.emplace(survivors[i]->task_id, PauseReason::AdmissionReconcile);
|
||||
}
|
||||
survivors.resize(static_cast<std::size_t>(std::max<std::int64_t>(global_cap, 0)));
|
||||
}
|
||||
|
||||
// ---- phase 3: fill free slots from Queued + governor-owned Paused ----------------
|
||||
std::int64_t global_running = static_cast<std::int64_t>(survivors.size());
|
||||
std::unordered_map<std::string, std::int64_t> per_queue_running;
|
||||
std::unordered_map<std::string, std::int64_t> per_host_running;
|
||||
for (const auto* t : survivors) {
|
||||
if (t->queue_id) ++per_queue_running[*t->queue_id];
|
||||
if (!t->host.empty()) ++per_host_running[t->host];
|
||||
}
|
||||
|
||||
auto resumable = [](const TaskView& t) {
|
||||
return t.run_state == RunState::Paused && t.pause_reason &&
|
||||
(*t.pause_reason == PauseReason::Schedule ||
|
||||
*t.pause_reason == PauseReason::QueueStopped ||
|
||||
*t.pause_reason == PauseReason::AdmissionReconcile);
|
||||
};
|
||||
|
||||
std::vector<const TaskView*> candidates;
|
||||
for (const auto& t : tasks) {
|
||||
if (t.run_state == RunState::Queued || resumable(t)) {
|
||||
if (queue_runnable(t.queue_id)) candidates.push_back(&t);
|
||||
}
|
||||
}
|
||||
std::sort(candidates.begin(), candidates.end(), less_priority);
|
||||
|
||||
std::vector<const TaskView*> newly_running;
|
||||
for (const auto* c : candidates) {
|
||||
if (global_running >= global_cap) break;
|
||||
if (c->queue_id && per_queue_running[*c->queue_id] >= queue_cap(*c->queue_id)) continue;
|
||||
if (!c->host.empty() && per_host_running[c->host] >= host_cap(c->host)) continue;
|
||||
|
||||
if (c->run_state == RunState::Queued) {
|
||||
d.to_start.push_back(c->task_id);
|
||||
} else {
|
||||
d.to_resume.push_back(c->task_id);
|
||||
}
|
||||
newly_running.push_back(c);
|
||||
++global_running;
|
||||
if (c->queue_id) ++per_queue_running[*c->queue_id];
|
||||
if (!c->host.empty()) ++per_host_running[c->host];
|
||||
}
|
||||
|
||||
// ---- priority order: everyone who will be running after this decision ------------
|
||||
std::vector<const TaskView*> will_run = survivors;
|
||||
will_run.insert(will_run.end(), newly_running.begin(), newly_running.end());
|
||||
std::sort(will_run.begin(), will_run.end(), less_priority);
|
||||
d.priority_order.reserve(will_run.size());
|
||||
for (const auto* t : will_run) d.priority_order.push_back(t->task_id);
|
||||
|
||||
return d;
|
||||
}
|
||||
|
||||
} // namespace velox::daemon::sched
|
||||
@@ -0,0 +1,89 @@
|
||||
#pragma once
|
||||
|
||||
// The concurrency governor — a pure decision function. Given a snapshot of every task's
|
||||
// coarse run-state and every queue's state, it decides which tasks should start, which
|
||||
// should resume, which should pause, and the priority order to push to CORE's segment
|
||||
// budget. No I/O, no store, no engine, no clock — the caller supplies an already-evaluated
|
||||
// world (schedule windows resolved via sched/schedule_window.hpp) and applies the result.
|
||||
//
|
||||
// Axes it enforces, all in TASK units (ADR 0011 §1: DAEMON counts tasks, CORE counts
|
||||
// segments):
|
||||
// * connection.maxConcurrentDownloads — global running-task ceiling
|
||||
// * min(that, connection.maxActiveSegments) — the one ADR 0011 §2 clamp
|
||||
// * Queue.maxConcurrent — per-queue running-task ceiling
|
||||
// * per-host task cap (== that host's segment cap) — ADR 0011 §4
|
||||
// * queue running state + schedule window — a stopped/closed queue runs nothing
|
||||
//
|
||||
// What it never does: touch a task paused for a reason it does not own (user, or CORE's
|
||||
// auto-pause for auth_required / server_file_changed / disk_full) — ADR 0013 §3.
|
||||
|
||||
#include <cstdint>
|
||||
#include <map>
|
||||
#include <optional>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
namespace velox::daemon::sched {
|
||||
|
||||
// The governor's coarse view of a task. Derived from the SQLite row + the engine's last
|
||||
// reported state; the governor does not need the fine-grained EngineState.
|
||||
enum class RunState {
|
||||
Queued, // wants a slot; eligible for admission
|
||||
Running, // probing / connecting / downloading / retry_wait / assembling / verifying
|
||||
Paused, // see pause_reason for whether the governor may resume it
|
||||
Terminal, // complete / failed / cancelled — ignored
|
||||
};
|
||||
|
||||
// Why a paused task is paused. Only the first three are governor-owned and auto-resumable;
|
||||
// `User` and `Auto` (CORE's auth/decision/disk auto-pause) are never touched here.
|
||||
enum class PauseReason { User, Schedule, QueueStopped, AdmissionReconcile, Auto };
|
||||
|
||||
struct TaskView {
|
||||
std::string task_id; // wire UUID
|
||||
RunState run_state = RunState::Queued;
|
||||
std::optional<PauseReason> pause_reason; // set iff run_state == Paused
|
||||
std::optional<std::string> queue_id; // nullopt => the main list (no queue)
|
||||
std::int64_t queue_position = 0; // order within the queue
|
||||
std::string host; // for the per-host cap; "" opts out
|
||||
std::int64_t admit_rank = 0; // FIFO tiebreak (created_at, then rowid)
|
||||
};
|
||||
|
||||
struct QueueView {
|
||||
std::string queue_id;
|
||||
bool running = false; // Queue.state == "running"
|
||||
std::int64_t max_concurrent = 1;
|
||||
bool window_open = true; // sched/schedule_window.hpp result for right now
|
||||
};
|
||||
|
||||
struct GovernorConfig {
|
||||
std::int64_t max_concurrent_downloads = 5; // connection.maxConcurrentDownloads
|
||||
std::int64_t max_active_segments = 32; // connection.maxActiveSegments (§2 clamp)
|
||||
std::map<std::string, std::int64_t> host_caps{}; // host -> max running tasks; absent => unlimited
|
||||
};
|
||||
|
||||
struct Decision {
|
||||
std::vector<std::string> to_start; // Queued -> admit: caller invokes Engine::start()
|
||||
std::vector<std::string> to_resume; // Paused (governor-owned reason) -> Engine::resume()
|
||||
std::vector<std::string> to_pause; // Running -> Engine::pause(); reason in pause_reasons
|
||||
std::map<std::string, PauseReason> pause_reasons; // task_id -> why, for the DB column
|
||||
std::vector<std::string> priority_order; // running + starting + resuming, for set_task_order()
|
||||
};
|
||||
|
||||
class Governor {
|
||||
public:
|
||||
Governor() = default;
|
||||
explicit Governor(GovernorConfig cfg) : cfg_(std::move(cfg)) {}
|
||||
|
||||
void set_config(GovernorConfig cfg) { cfg_ = std::move(cfg); }
|
||||
const GovernorConfig& config() const noexcept { return cfg_; }
|
||||
|
||||
// Pure: same inputs -> same Decision. `tasks` and `queues` are snapshots; order within
|
||||
// them does not matter (the governor sorts by its own keys).
|
||||
Decision evaluate(const std::vector<TaskView>& tasks,
|
||||
const std::vector<QueueView>& queues) const;
|
||||
|
||||
private:
|
||||
GovernorConfig cfg_;
|
||||
};
|
||||
|
||||
} // namespace velox::daemon::sched
|
||||
@@ -0,0 +1,89 @@
|
||||
#include "sched/schedule_window.hpp"
|
||||
|
||||
#include <array>
|
||||
#include <charconv>
|
||||
#include <optional>
|
||||
#include <string_view>
|
||||
|
||||
namespace velox::daemon::sched {
|
||||
|
||||
namespace proto = velox::proto;
|
||||
|
||||
namespace {
|
||||
|
||||
// "HH:MM" -> minutes since midnight. Returns nullopt on a malformed value (the schema
|
||||
// pattern should prevent that, but the schedule can also come from an older DB row).
|
||||
std::optional<int> minutes_of(std::string_view hhmm) {
|
||||
if (hhmm.size() != 5 || hhmm[2] != ':') return std::nullopt;
|
||||
int h = 0;
|
||||
int m = 0;
|
||||
if (std::from_chars(hhmm.data(), hhmm.data() + 2, h).ec != std::errc{}) return std::nullopt;
|
||||
if (std::from_chars(hhmm.data() + 3, hhmm.data() + 5, m).ec != std::errc{}) return std::nullopt;
|
||||
if (h < 0 || h > 23 || m < 0 || m > 59) return std::nullopt;
|
||||
return h * 60 + m;
|
||||
}
|
||||
|
||||
bool in_window(int now_min, std::optional<int> start_min, std::optional<int> stop_min) {
|
||||
const int start = start_min.value_or(0);
|
||||
if (!stop_min) return now_min >= start; // null stop => until end of day
|
||||
const int stop = *stop_min;
|
||||
if (start <= stop) return now_min >= start && now_min < stop;
|
||||
return now_min >= start || now_min < stop; // overnight window
|
||||
}
|
||||
|
||||
// tm_year/mon/mday -> "YYYY-MM-DD" to compare against onceDate.
|
||||
std::array<char, 11> iso_date(const std::tm& t) {
|
||||
std::array<char, 11> out{};
|
||||
const int y = t.tm_year + 1900;
|
||||
const int mo = t.tm_mon + 1;
|
||||
const int d = t.tm_mday;
|
||||
out[0] = static_cast<char>('0' + (y / 1000) % 10);
|
||||
out[1] = static_cast<char>('0' + (y / 100) % 10);
|
||||
out[2] = static_cast<char>('0' + (y / 10) % 10);
|
||||
out[3] = static_cast<char>('0' + y % 10);
|
||||
out[4] = '-';
|
||||
out[5] = static_cast<char>('0' + mo / 10);
|
||||
out[6] = static_cast<char>('0' + mo % 10);
|
||||
out[7] = '-';
|
||||
out[8] = static_cast<char>('0' + d / 10);
|
||||
out[9] = static_cast<char>('0' + d % 10);
|
||||
out[10] = '\0';
|
||||
return out;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool window_open(const proto::Schedule& s, const std::tm& now) {
|
||||
if (!s.enabled) return true;
|
||||
|
||||
const int now_min = now.tm_hour * 60 + now.tm_min;
|
||||
const std::optional<int> start = s.startTime ? minutes_of(*s.startTime) : std::nullopt;
|
||||
const std::optional<int> stop = s.stopTime ? minutes_of(*s.stopTime) : std::nullopt;
|
||||
|
||||
if (s.mode == proto::ScheduleMode::Once) {
|
||||
if (!s.onceDate) return false; // "once" with no date never runs
|
||||
const auto today = iso_date(now);
|
||||
if (std::string_view(today.data()) != *s.onceDate) return false;
|
||||
return in_window(now_min, start, stop);
|
||||
}
|
||||
|
||||
// periodic
|
||||
if (s.daysOfWeek && !s.daysOfWeek->empty()) {
|
||||
bool day_ok = false;
|
||||
for (const auto d : *s.daysOfWeek) {
|
||||
if (d == now.tm_wday) {
|
||||
day_ok = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!day_ok) return false;
|
||||
}
|
||||
return in_window(now_min, start, stop);
|
||||
}
|
||||
|
||||
bool window_open(const std::optional<proto::Schedule>& s, const std::tm& now) {
|
||||
if (!s) return true;
|
||||
return window_open(*s, now);
|
||||
}
|
||||
|
||||
} // namespace velox::daemon::sched
|
||||
@@ -0,0 +1,33 @@
|
||||
#pragma once
|
||||
|
||||
// Evaluates a queue's Schedule against the current local wall-clock time: is the queue
|
||||
// allowed to be running right now? Times are local HH:MM and re-evaluated on every tick
|
||||
// (the daemon never caches an absolute instant — docs, Schedule.schema.json), so this is
|
||||
// a pure function of (schedule, broken-down local time).
|
||||
//
|
||||
// A disabled schedule, or no schedule at all, means "always open" — the queue runs
|
||||
// whenever queue.state is 'running'.
|
||||
|
||||
#include <ctime>
|
||||
|
||||
#include "velox_proto.hpp"
|
||||
|
||||
namespace velox::daemon::sched {
|
||||
|
||||
// `local_now` must be a fully populated std::tm in local time (tm_wday and tm_year/mon/mday
|
||||
// used). Returns true when the schedule permits the queue to run at that moment.
|
||||
//
|
||||
// - mode "once": open iff the date matches onceDate and the time is within the window.
|
||||
// daysOfWeek is ignored.
|
||||
// - mode "periodic": open iff today's weekday is in daysOfWeek (an empty/absent list means
|
||||
// every day) and the time is within the window.
|
||||
//
|
||||
// The window is [startTime, stopTime). A null startTime means 00:00. A null stopTime
|
||||
// means "until end of day" (and, for a queue, "until it drains"). A stopTime earlier than
|
||||
// startTime is an overnight window (e.g. 22:00-06:00): open if now >= start OR now < stop.
|
||||
bool window_open(const velox::proto::Schedule& schedule, const std::tm& local_now);
|
||||
|
||||
// Convenience: the queue's schedule is optional. nullopt or disabled => always open.
|
||||
bool window_open(const std::optional<velox::proto::Schedule>& schedule, const std::tm& local_now);
|
||||
|
||||
} // namespace velox::daemon::sched
|
||||
Reference in New Issue
Block a user