optimize the below solution for maximum score Jud...

Créé le : 14 août 2026

Répondu en utilisant GPT-5.6 Thinking par Chat01

Question

image.png
image.png
image.png
image.png
image.png

optimize the below solution for maximum score

Judge Protocol
#1: OK [0ms, 0MB]: points 500.0000027586 tp=0.022222 mean_tdr=30.000000 mean_tpot=0.000000 dist=0.000000 norm_tp=0.000000 norm_c=1.000000 normalized_score=0.500000 points=500.000003
#2: OK [0ms, 0MB]: points 500.0 tp=0.005755 mean_tdr=126.158679 mean_tpot=0.000000 dist=0.000000 norm_tp=0.000000 norm_c=1.000000 normalized_score=0.500000 points=500.000000
#3: OK [31ms, 0MB]: points 342.264433074 tp=0.004326 mean_tdr=1329.849832 mean_tpot=97.079554 dist=0.760859 norm_tp=0.563796 norm_c=0.342264 normalized_score=0.342264 points=342.264433
#4: OK [15ms, 0MB]: points 698.4700225801 tp=0.027968 mean_tdr=475.428598 mean_tpot=301.969724 dist=2.174659 norm_tp=0.181822 norm_c=0.919890 normalized_score=0.698470 points=698.470023
#5: OK [796ms, 0MB]: points 218.786393653 tp=0.105429 mean_tdr=1888.825861 mean_tpot=1106.189262 dist=10.621745 norm_tp=0.025051 norm_c=0.993729 normalized_score=0.218786 points=218.786394
#6: OK [437ms, 0MB]: points 124.750812372 tp=0.084160 mean_tdr=3521.755694 mean_tpot=845.218630 dist=13.514000 norm_tp=0.029822 norm_c=0.979111 normalized_score=0.124751 points=124.750812
#7: OK [0ms, 0MB]: points 906.3653187913 tp=0.014362 mean_tdr=860.122989 mean_tpot=68.298854 dist=0.376199 norm_tp=0.418177 norm_c=0.906365 normalized_score=0.906365 points=906.365319
#8: OK [15ms, 0MB]: points 759.683432113 tp=0.011025 mean_tdr=1086.885839 mean_tpot=224.450400 dist=2.038185 norm_tp=0.600486 norm_c=0.812749 normalized_score=0.759683 points=759.683432
#9: OK [31ms, 0MB]: points 666.4087338479 tp=0.003820 mean_tdr=6920.189321 mean_tpot=0.000000 dist=11.490623 norm_tp=0.777425 norm_c=0.660566 normalized_score=0.666409 points=666.408734
#10: OK [125ms, 0MB]: points 569.4053617673 tp=0.006204 mean_tdr=182687.789798 mean_tpot=6651.683747 dist=176.313250 norm_tp=0.698551 norm_c=0.546615 normalized_score=0.569405 points=569.405362
#11: OK [15ms, 0MB]: points 500.185382213 tp=0.000007 mean_tdr=32780482.884393 mean_tpot=16199.089335 dist=0.000000 norm_tp=0.000371 norm_c=1.000000 normalized_score=0.500185 points=500.185382
#12: OK [31ms, 0MB]: points 803.2244568309 tp=0.000024 mean_tdr=1382425.696096 mean_tpot=484.825909 dist=3.630814 norm_tp=0.809404 norm_c=0.191410 normalized_score=0.803224 points=803.224457
#13: OK [15ms, 0MB]: points 478.2292859762 tp=0.018901 mean_tdr=1698.624971 mean_tpot=293.229720 dist=4.919113 norm_tp=0.401396 norm_c=0.708730 normalized_score=0.478229 points=478.229286
#14: OK [46ms, 0MB]: points 415.2668658781 tp=0.003564 mean_tdr=192.489397 mean_tpot=184.378198 dist=0.176642 norm_tp=0.210323 norm_c=0.795876 normalized_score=0.415267 points=415.266866
#15: OK [15ms, 0MB]: points 714.5489915731 tp=0.000009 mean_tdr=19314480.365600 mean_tpot=0.000000 dist=90.995602 norm_tp=0.982404 norm_c=0.495395 normalized_score=0.714549 points=714.548992
#16: OK [31ms, 0MB]: points 763.0636643412 tp=0.023740 mean_tdr=41971.720507 mean_tpot=2519.731209 dist=48.119562 norm_tp=0.760681 norm_c=0.879835 normalized_score=0.763064 points=763.063664
#17: OK [859ms, 0MB]: points 833.4604346598 tp=0.000520 mean_tdr=29116626.412248 mean_tpot=25241.134141 dist=1556.444052 norm_tp=0.986650 norm_c=0.522438 normalized_score=0.833460 points=833.460435
#18: OK [78ms, 0MB]: points 784.1173342212 tp=0.000009 mean_tdr=46957956.216119 mean_tpot=0.000000 dist=371.688376 norm_tp=0.991025 norm_c=0.498388 normalized_score=0.784117 points=784.117334
#19: OK [5234ms, 0MB]: points 1.9484116884 tp=0.015177 mean_tdr=32458.900838 mean_tpot=59685.455078 dist=1196.599265 norm_tp=0.001948 norm_c=0.971085 normalized_score=0.001948 points=1.948412
#20: OK [265ms, 0MB]: points 998.1826939349 tp=0.005607 mean_tdr=1245.882662 mean_tpot=172.801195 dist=0.000000 norm_tp=0.995218 norm_c=1.000000 normalized_score=0.998183 points=998.182694
#21: OK [62ms, 0MB]: points 836.9650121087 tp=0.010034 mean_tdr=69275.738940 mean_tpot=0.000000 dist=291.069280 norm_tp=0.773682 norm_c=0.900248 normalized_score=0.836965 points=836.965012
#22: OK [4453ms, 0MB]: points 497.1021540159 tp=0.248843 mean_tdr=2782.002157 mean_tpot=3707.999990 dist=616.473451 norm_tp=0.001910 norm_c=0.992295 normalized_score=0.497102 points=497.102154
→ SourceCopy
#include <bits/stdc++.h>
using namespace std;

enum class State {
NONE,

text
P_PRE_READY, P_PRE_RUN, WAIT_P_UP, P_PROC_READY, P_PROC_RUN, WAIT_P_DOWN, P_POST_READY, P_POST_RUN, D_PRE_READY, D_PRE_RUN, WAIT_D_UP, D_PROC_READY, D_PROC_RUN, WAIT_D_DOWN, D_POST_READY, D_POST_RUN, FINISHED

};

struct Request {
int lin = 0;
int remote = -1;
State st = State::NONE;
bool finished = false;
};

struct ReadyItem {
int rid;
State expected;
};

int main() {
ios::sync_with_stdio(false);
cin.tie(nullptr);

text
// ------------------------------------------------------------ // Startup configuration // ------------------------------------------------------------ int K; double S; double latency_in_ms; double bandwidth_gbps; long long bytes_per_token; int num_layers; if (!(cin >> K >> S >> latency_in_ms >> bandwidth_gbps >> bytes_per_token >> num_layers)) { return 0; } // Scoring parameters. A correctness-first scheduler does not // need to use them, but they must be read. double SLO1, SLO2; double tp_UB, tp_base, dist_base, w_tp, w_c; cin >> SLO1 >> SLO2 >> tp_UB >> tp_base >> dist_base >> w_tp >> w_c; // ------------------------------------------------------------ // Task-time table // ------------------------------------------------------------ int N; cin >> N; for (int i = 0; i < N; ++i) { int batch_size; double prefill_pre; double prefill_proc; double prefill_post; double decode_pre; double decode_proc; double decode_post; cin >> batch_size >> prefill_pre >> prefill_proc >> prefill_post >> decode_pre >> decode_proc >> decode_post; } // ------------------------------------------------------------ // Scheduler state // ------------------------------------------------------------ vector<Request> req; // True iff the corresponding computer currently has an // assigned task whose TDN has not yet arrived. bool edgeBusy = false; vector<bool> cloudBusy(K, false); // Number of unfinished requests currently assigned to each cloud. // Used only to balance new P PRE assignments. vector<int> activeOnCloud(K, 0); // Tasks ready to run on the edge. deque<ReadyItem> edgeReady; // Tasks ready to run on each cloud. vector<deque<ReadyItem>> cloudReady(K); auto ensureRequest = [&](int rid) { if ((int)req.size() <= rid) req.resize(rid + 1); }; auto pushEdge = [&](int rid, State s) { edgeReady.push_back({rid, s}); }; auto pushCloud = [&](int remote, int rid, State s) { cloudReady[remote].push_back({rid, s}); }; // A queue can contain a stale entry, most notably when the final // D POST TDN and FIN occur in the same frame. auto edgeItemValid = [&](const ReadyItem& x) -> bool { if (x.rid < 0 || x.rid >= (int)req.size()) return false; const Request& r = req[x.rid]; return !r.finished && r.st == x.expected; }; auto cloudItemValid = [&](int cloud, const ReadyItem& x) -> bool { if (x.rid < 0 || x.rid >= (int)req.size()) return false; const Request& r = req[x.rid]; return !r.finished && r.remote == cloud && r.st == x.expected; }; // Pick a cloud for a new request. // Output length is hidden, so balancing the number of live // requests is a simple robust rule. auto chooseCloud = [&]() -> int { int best = 0; for (int k = 1; k < K; ++k) { if (activeOnCloud[k] < activeOnCloud[best]) { best = k; } else if (activeOnCloud[k] == activeOnCloud[best]) { // Small secondary preference for an idle cloud. if (cloudBusy[best] && !cloudBusy[k]) { best = k; } else if (cloudBusy[k] == cloudBusy[best]) { // Then prefer the shorter currently-ready queue. if (cloudReady[k].size() < cloudReady[best].size()) best = k; } } } return best; }; // ------------------------------------------------------------ // Interactive frame loop // ------------------------------------------------------------ while (true) { string firstToken; // EOF / interactor closure => exit cleanly. if (!(cin >> firstToken)) return 0; if (firstToken == "END") return 0; // Timestamp is deliberately not needed by this reactive // scheduler. // Reading as a string also avoids unnecessary floating point work. string timestamp = firstToken; int eventCount; if (!(cin >> eventCount)) return 0; // -------------------------------------------------------- // Read and process the WHOLE frame before scheduling. // -------------------------------------------------------- for (int ev = 0; ev < eventCount; ++ev) { string type; cin >> type; // ==================================================== // ARR // ==================================================== if (type == "ARR") { int rid, lin; cin >> rid >> lin; ensureRequest(rid); req[rid].lin = lin; req[rid].remote = -1; req[rid].finished = false; req[rid].st = State::P_PRE_READY; pushEdge(rid, State::P_PRE_READY); } // ==================================================== // TDN // ==================================================== else if (type == "TDN") { string server; string family; string step; cin >> server >> family >> step; // Resource becomes free immediately for this frame. if (server == "E") { edgeBusy = false; } else { // Server format: C0, C1, ... int k = stoi(server.substr(1)); cloudBusy[k] = false; } // ------------------------------------------------ // Prefill / input stage // ------------------------------------------------ if (family == "P") { if (step == "PRE") { int remote, rid; double dur; cin >> remote >> rid >> dur; ensureRequest(rid); if (!req[rid].finished) { // P PRE completion automatically queues UP. req[rid].st = State::WAIT_P_UP; } } else if (step == "PROC") { int ls, le, remote, rid; double dur; cin >> ls >> le >> remote >> rid >> dur; ensureRequest(rid); if (!req[rid].finished) { // This solution always uses the full final // piece [0, num_layers), so completion queues DOWN. req[rid].st = State::WAIT_P_DOWN; } } else if (step == "POST") { int remote, rid; double dur; cin >> remote >> rid >> dur; ensureRequest(rid); if (!req[rid].finished) { // First decode iteration is now ready. req[rid].st = State::D_PRE_READY; pushEdge(rid, State::D_PRE_READY); } } } // ------------------------------------------------ // Decode / output stage // ------------------------------------------------ else if (family == "D") { if (step == "PRE") { int marker, m; cin >> marker >> m; vector<int> ids(m); for (int& rid : ids) cin >> rid; double dur; cin >> dur; for (int rid : ids) { ensureRequest(rid); if (!req[rid].finished) { // D PRE completion automatically queues UP. req[rid].st = State::WAIT_D_UP; } } } else if (step == "PROC") { int remote, m; cin >> remote >> m; vector<int> ids(m); for (int& rid : ids) cin >> rid; double dur; cin >> dur; for (int rid : ids) { ensureRequest(rid); if (!req[rid].finished) { // D PROC completion automatically queues DOWN. req[rid].st = State::WAIT_D_DOWN; } } } else if (step == "POST") { int marker, m; cin >> marker >> m; vector<int> ids(m); for (int& rid : ids) cin >> rid; double dur; cin >> dur; for (int rid : ids) { ensureRequest(rid); /* * If this was NOT the final token, its next * D PRE becomes ready. * * If it WAS the final token, FIN is in this * same frame. * * If FIN occurs after this line, the queued * entry becomes stale and is ignored later. * * If FIN occurred before this line, the * finished check prevents resurrecting it. */ if (!req[rid].finished) { req[rid].st = State::D_PRE_READY; pushEdge(rid, State::D_PRE_READY); } } } } } // ==================================================== // XDN // ==================================================== else if (type == "XDN") { string direction; int remote; long long sizeBytes; string phase; int m; cin >> direction >> remote >> sizeBytes >> phase >> m; vector<int> ids(m); for (int& rid : ids) cin >> rid; // ------------------------------------------------ // Prefill transfer // ------------------------------------------------ if (phase == "PRE") { for (int rid : ids) { ensureRequest(rid); if (req[rid].finished) continue; if (direction == "UP") { // Input arrived at its chosen cloud. req[rid].st = State::P_PROC_READY; pushCloud(remote, rid, State::P_PROC_READY); } else { // Input-stage result arrived back at edge. req[rid].st = State::P_POST_READY; pushEdge(rid, State::P_POST_READY); } } } // ------------------------------------------------ // Decode transfer // ------------------------------------------------ else { // DEC for (int rid : ids) { ensureRequest(rid); if (req[rid].finished) continue; if (direction == "UP") { req[rid].st = State::D_PROC_READY; pushCloud(remote, rid, State::D_PROC_READY); } else { req[rid].st = State::D_POST_READY; pushEdge(rid, State::D_POST_READY); } } } } // ==================================================== // FIN // ==================================================== else if (type == "FIN") { int rid; cin >> rid; ensureRequest(rid); if (!req[rid].finished) { req[rid].finished = true; req[rid].st = State::FINISHED; int k = req[rid].remote; if (0 <= k && k < K) { --activeOnCloud[k]; } } } } // -------------------------------------------------------- // Frame fully read. // // Now choose assignments. // // No assignment below is allowed to depend on another // assignment from this same response. // -------------------------------------------------------- vector<string> answer; answer.reserve(K + 1); // ======================================================== // Schedule the EDGE // ======================================================== if (!edgeBusy) { while (!edgeReady.empty() && !edgeItemValid(edgeReady.front())) { edgeReady.pop_front(); } if (!edgeReady.empty()) { ReadyItem item = edgeReady.front(); edgeReady.pop_front(); int rid = item.rid; Request& r = req[rid]; if (r.st == State::P_PRE_READY) { int k = chooseCloud(); r.remote = k; r.st = State::P_PRE_RUN; ++activeOnCloud[k]; answer.push_back( "E P PRE " + to_string(k) + " " + to_string(rid) ); edgeBusy = true; } else if (r.st == State::P_POST_READY) { r.st = State::P_POST_RUN; answer.push_back( "E P POST " + to_string(r.remote) + " " + to_string(rid) ); edgeBusy = true; } else if (r.st == State::D_PRE_READY) { r.st = State::D_PRE_RUN; // Singleton decode group. answer.push_back( "E D PRE -1 1 " + to_string(rid) ); edgeBusy = true; } else if (r.st == State::D_POST_READY) { r.st = State::D_POST_RUN; // Singleton decode group. answer.push_back( "E D POST -1 1 " + to_string(rid) ); edgeBusy = true; } } } // ======================================================== // Schedule each CLOUD independently // ======================================================== for (int k = 0; k < K; ++k) { if (cloudBusy[k]) continue; while (!cloudReady[k].empty() && !cloudItemValid(k, cloudReady[k].front())) { cloudReady[k].pop_front(); } if (cloudReady[k].empty()) continue; ReadyItem item = cloudReady[k].front(); cloudReady[k].pop_front(); int rid = item.rid; Request& r = req[rid]; if (r.st == State::P_PROC_READY) { r.st = State::P_PROC_RUN; // One full, ascending, gap-free prefill piece. answer.push_back( "C" + to_string(k) + " P PROC 0 " + to_string(num_layers) + " " + to_string(k) + " " + to_string(rid) ); cloudBusy[k] = true; } else if (r.st == State::D_PROC_READY) { r.st = State::D_PROC_RUN; // Singleton cloud-local decode group. answer.push_back( "C" + to_string(k) + " D PROC " + to_string(k) + " 1 " + to_string(rid) ); cloudBusy[k] = true; } } // -------------------------------------------------------- // Send exactly one response for this frame. // -------------------------------------------------------- cout << answer.size() << '\n'; for (const string& line : answer) cout << line << '\n'; cout << flush; } return 0;

}A. Edge–Cloud Collaborative Scheduling
time limit per test15 seconds
memory limit per test256 megabytes
This is an interactive scheduling problem. The "interactor" sends events as they happen; after each event group, you choose ready tasks to start. You never predict future requests. For a fixed test it is deterministic and non-adaptive: the same legal responses produce the same events and score.

No AI Background Is Needed

Treat this as jobs moving through computers; no text-generation knowledge is needed. A token is one output unit. Commands P PRE/P PROC/P POST prepare a request once; D PRE/D PROC/D POST produce one token. Both follow local →
remote →
local.

Term Meaning
edge / local the one computer beside users
cloud / remote one of 𝐾
worker computers
token one output unit
prefill / input stage one-time input preparation
decode / output step repeated work producing one token
uplink / UP local-to-remote transfer
downlink / DOWN remote-to-local transfer
FIFO queue first transfer queued is the first transfer completed
batch / group requests combined in one output task
chunk / piece a consecutive range of input-stage parts
activations transferred request data; its size is defined by the protocol
model layers / parts the numbered input-stage range [0,num_layers)
The Problem in One Minute

One local computer and several remote computers handle requests. Each request first prepares its input by making this trip:

local computer⟶remote computer⟶local computer.

The same trip is then repeated once per token. The first trip is the input stage; each later trip is an output step. Transfers are automatic; you choose ready tasks for free computers.

The score rewards output rate and short waits. Time to decode ready (TDR) ends when the first output step can begin, before a token is produced. Time per output token (TPOT) is the mean gap between consecutive tokens.

Challenge Format

Two worked examples appear at the end of the statement. Test 1 is the public worked Example 1, tests 2–22 are hidden preliminary tests, and finals use a separate frozen set. Per-test scores are reported on the 0–1000 scale defined below, and the contest system aggregates them according to the contest rules; no hacks are used.

Guide: blue is local, green is the shared two-way link, and orange is remote; computation and transfer may overlap. Figure labels use the glossary above.

What You Control

You choose:

which legal task to start on each free computer;
the remote computer assigned to a request when scheduling its P PRE;
how to split an input-stage P PROC into ranges of numbered parts; and
which ready requests to combine into each output group.
Transfers are automatic. For a first solution, use one full input-stage piece and groups of size 1
.

System Model

There is one local computer and 𝐾
identical remote computers, numbered 0
through 𝐾−1
and written C0, C1, …
. They work independently and may run while data moves.
There are 𝑅
requests in total, but 𝑅
is not announced. Request ids are 𝑖=0,1,…,𝑅−1
in arrival order. The arrival time of request 𝑖
is the timestamp of its ARR event. That event reveals its input length 𝐿in[𝑖]
; its output length 𝐿out[𝑖]
remains hidden until a FIN event reports that it has finished.
Request 𝑖
performs one token-free input stage, then exactly 𝐿out[𝑖]
output steps, each producing one token. Thus total generated tokens are ∑𝑖𝐿out[𝑖]
, with 𝐿out[𝑖]≥1
.
You assign each request to one remote computer in its P PRE task. The choice is fixed: all later input stage tasks for that request name the same remote computer, and every D PROC containing it runs there. Output-step tasks on the local computer may combine requests from different remote computers.
Each computer executes at most one task at a time. Once a task starts, it cannot be paused. Separately scheduled input-stage pieces may be alternated with other work.
Schedule cost 𝑆
: a task assigned at time 𝑡
with execution duration dur
occupies its computer over [𝑡,𝑡+𝑆+dur]
. The cost is paid once per task, including once per input-stage piece and once per output group. Transfers do not pay 𝑆
.
Execution times are fixed: task duration depends on its step and group size (and fraction of parts for input-stage pieces), never on the output-token position or saved internal data size. See Input — Task-Time Table. Finished tasks and transfers are reported as TDN and XDN.
Request Lifecycle

Solid arrows are scheduling dependencies; dashed arrows are automatic transfers.

Guide: TDR ends at P POST; each later D POST makes one token.

PRE/POST use the local computer; PROC uses the assigned remote; transfers use the shared link.

Input Stage

Three steps: first (local) →
process (assigned remote) →
final (local). An input-stage group has one request.

P PRE (local computer). May start after: the request's ARR event. Fixes the remote computer; on completion, the local-to-remote transfer is queued immediately (whether or not that remote computer is busy).
P PROC (remote computer) computes all num_layers
numbered parts. The simplest choice is one full piece [0,num_layers)
. You may split it so other remote work can run between pieces.
P POST (local computer). May start after: the request's input stage remote-to-local transfer XDN. Completion makes the request ready for producing output; TDR is measured from arrival to this completion.
Exact splitting rules. Each piece is a nonempty integer range [𝑙𝑠,𝑙𝑒)
: it includes part 𝑙𝑠
and stops before part 𝑙𝑒
. For each request, pieces must be issued in ascending, gap-free order: the first starts at 0
, each later 𝑙𝑠
equals the previous 𝑙𝑒
, and the last ends at num_layers
. Thus there are at most num_layers
pieces, and none when num_layers=1
beyond the full piece. The first piece waits for the input stage local-to-remote transfer XDN; each later piece waits for the previous piece's TDN. Its duration is

𝑙𝑒−𝑙𝑠num_layers×prefill_proc(𝐿in[𝑖]).

Only the last piece queues the input stage remote-to-local transfer, of length 𝐿in[𝑖]
. No transfer occurs between pieces. Any range that is empty, outside the part interval, out of order, or leaves a gap is a violation.

Output Steps

After the input stage, request 𝑖
performs 𝐿out[𝑖]
output steps in succession: completing iteration 𝑘
's final step readies iteration 𝑘+1
. The same three steps are used, but multiple requests may be grouped per step. You choose each group independently. Output-step tasks cannot be split into part ranges.

D PRE (local computer, across remote computers): may group requests assigned to any mix of remote computers — the listed decode_pre
time depends only on total group size, so across remote computers grouping improves local computer efficiency (the task-time table plus the transfer formula is the only efficiency model; nothing else rewards grouping). May start after, for each request: completion of that request's previous final step (P POST for its first output step, otherwise its previous D POST) — and the request must not be finished (FIN always arrives with the final D POST TDN, so you always know). Completion queues one local-to-remote transfer per distinct remote computer in the group (len =
that remote computer's member count), enqueued simultaneously in increasing remote computer index order.
D PROC (remote computer): all members must be assigned to that remote computer. May start after, for each request: the local-to-remote transfer XDN carrying that member's current-iteration data — the rule is checked separately for each request: members may come from different D PRE groups and different local-to-remote transfers. Completion queues one remote-to-local transfer (len =
this group's size).
D POST (local computer, across remote computers). May start after, for each request: the remote-to-local transfer XDN carrying that member's current-iteration result (different D PROC groups/remote-to-local transfers are fine). Each member's completion produces one token.

Guide: the upper lane splits one input stage into consecutive ranges; the lower lane groups ready requests for output.

Groups: 𝑚≥1
, request ids distinct, every member satisfying its predecessor rule; 𝑚=1
is always allowed. Only output work is ever grouped — input-stage groups always contain exactly one request. There is no other group-size limit: no maximum exists or is announced — any nonempty set of distinct, currently ready requests may be grouped, subject only to the step and remote computer rules above. Order of request ids within a group: see Output.

Input
Protocol Guide

P is the one-time input stage; D is one output step. PRE/POST are local tasks and PROC is remote. Events are ARR (arrival), TDN (task done), XDN (transfer done), and FIN (finished). In [𝑙𝑠,𝑙𝑒)
, 𝑙𝑠
is included and 𝑙𝑒
excluded. Input arrives interactively: read each whole frame and respond once. Times are real milliseconds; counts, lengths, and ids are integers; fields use single spaces.

Startup Configuration

The interactor first sends two lines (no response expected): the system parameters, then the scoring parameters:

K S latency_in_ms bandwidth_gbps bytes_per_token num_layers
SLO1 SLO2 tp_UB tp_base dist_base w_tp w_c
𝐾
, bytes_per_token
, and num_layers
are integers; all other values are reals (times in ms, output rates in tokens/ms). SLO stands for service-level objective: SLO1 is the TDR target and SLO2 is the TPOT target. The other names on the scoring line mean: tp_UB =𝑡𝑝UB
, tp_base =𝑡𝑝base
, dist_base =𝑑𝑖𝑠𝑡base
, w_tp =𝑤tp
, w_c =𝑤𝑐
. bandwidth_gbps
is in gigabits per second (Gb/s). The value bytes_per_token
is already the complete data size for one token; do not multiply it by an element width.

Communication

Transfers are automatic: the interactor queues them when their triggering computation finishes; you never output a transfer command. A single link connects the local computer and remote computer side and is shared by all remote computers. UP means a local-to-remote transfer (local computer →
remote computer), and DOWN means a remote-to-local transfer (remote computer →
local computer). These are independent one-at-a-time FIFO queues: transfers finish in the order they enter, and both directions may be active simultaneously. For simultaneous queues, one D PRE's per-remote transfers enter by increasing remote index. Otherwise transfers enter in interactor event order; tasks started together and finishing together use assignment-line order.

Transfer time =latency_in_ms+8data_bytes/(bandwidth_gbps×106)
ms, where data_bytes=len×bytes_per_token
. The factor 8
converts bytes to bits. For an input-stage transfer of request 𝑖
, len=𝐿in[𝑖]
; for output, use the per-remote computer local-to-remote transfer and per-group remote-to-local transfer sizes defined in Legend. Guaranteed latency_in_ms>0
: every transfer takes strictly positive time.

Task-Time Table

Before requests arrive, the judge gives you a table telling you how long tasks take, in milliseconds. You only read this table; you do not measure the times yourself. Read an integer 𝑁
, then 𝑁
rows of 7 values; no response is expected:

batch_size prefill_pre prefill_proc prefill_post decode_pre decode_proc decode_post
Here batch_size means the number of requests grouped into one task. The names containing prefill refer to the input stage, and the names containing decode refer to output steps. These fixed names come from the protocol; no AI knowledge is required. The values help you compare scheduling choices. You must read all rows, but a first correct scheduler does not need to use the values at all; a more advanced scheduler will use the complete task-time table, including values between listed rows.

batch_size means 𝐿in
for the three input stage columns (one request per input stage group) and member count for the three output columns. The batch_size values are distinct positive integers in [1,4096]
. A listed output-step group size may exceed the test's 𝑅
; such a row only defines the task-time table and does not make that group size schedulable. Not every step is listed at every group size; a missing value is -1. Rows are given in no guaranteed order. Guarantee: every step column contains at least one non-missing entry.

All listed values are execution durations excluding the schedule cost 𝑆
, exactly like the dur field of every TDN: the total time for which a computer is busy with a task is always 𝑆+dur
(see System Model). For each step, sort the available rows by group size. If the needed size is listed, use its time. If it lies between two listed sizes, draw a straight line between their times and use the value on that line. Below the smallest size, use the first time; above the largest size, use the last time. The result is the same on every remote computer and never changes. Guarantee: every resulting legal task duration is strictly positive. Here 𝑅
is the test's total request count (see Constraints); 𝑅
is not announced during the interaction and is not a separate group-size limit. No maximum output group size exists beyond the number of currently ready requests. The duration of every completed task is echoed in its TDN's dur field.

Event Frames (Your Turns)

The interactor sends a frame whenever one or more events occur. A frame contains: one timestamp line t, one event-count line e, then 𝑒
event lines. Always read the whole frame before deciding what to do.

The frame timestamp is its event time, printed with 9 decimal places; printed timestamps are nondecreasing, while the internal event times of consecutive frames are strictly increasing. Only events with exactly the same internal timestamp are coalesced. A later event is never moved to an earlier frame, and line order carries no scheduling priority. Your response may use any event in the frame: every resource freed by a TDN is free, even if that event is the last line. Guarantee: FIN appears beside the TDN of that request's final output final step. After reading such a frame, the request is finished and must not appear in your response to that frame or any later response.

Guide: read the whole event frame, update state, then print one count and exactly that many assignments.

ARR <rid> <𝐿in[𝑖]

— request 𝑖=𝐫𝐢𝐝
arrived; its arrival time is this frame's timestamp 𝑡
; 𝐿in[𝑖]
is given, while 𝐿out[𝑖]
is unknown.
TDN <server> <task_spec> <dur> — a task completed and its server is now free. task_spec is echoed in a canonical equivalent form: integers use ordinary decimal notation and fields use single spaces; dur is the execution duration, excluding 𝑆
, printed with 9 digits after the decimal point.
XDN <UP|DOWN> <remote> <size> <PRE|DEC> <m> <rid...> — a transfer completed (its data has arrived). size is in bytes (=len×bytes_per_token
); PRE/DEC marks an input-stage or output-step transfer; per-remote computer transfers each produce their own XDN; for input stage, m =1
. The m request ids listed are exactly the requests whose data the transfer carries: the single request for input stage; that remote computer's members of the triggering D PRE for an output-step local-to-remote transfer; the members of the triggering D PROC group for an output-step remote-to-local transfer.
FIN <rid> — the request finished all output steps (its last token was just produced). It must not appear in any task you assign from now on.
<server> =
E (local computer) or Ck (remote computer 𝑘
); <m> is the number of request ids that follow. request ids are integers 0,1,…,𝑅−1
, assigned in arrival order, never reused.

Constraints

1≤𝐾≤8
; 1≤𝑅≤2000
; 1≤𝐿in[𝑖]≤4096
; 1≤𝐿out[𝑖]≤512
; ∑𝑖𝐿out[𝑖]≤2⋅105
per test; 1≤num_layers≤64
; 2≤𝑁≤4096
.
1≤𝑆≤10
; 0.001≤latency_in_ms≤50
; 0.001≤bandwidth_gbps≤100
; 1≤bytes_per_token≤106
.
0.001≤SLO1,SLO2≤109
; 0≤𝑡𝑝base,𝑑𝑖𝑠𝑡base≤109
; 10−9≤𝑡𝑝UB≤109
; and 𝑡𝑝UB>𝑡𝑝base
.
Every non-missing task-table entry is in [0.001,104]
ms. System reals, task times, timestamps, and durations use 9 decimal places.
Arrival timestamps are nondecreasing in [0,109]
ms. Completion frames may exceed 109
, but the validator guarantees a conservative upper bound of 1012
ms even if every legal task and transfer is serialized; all timestamps are nonnegative finite doubles.
You never need to predict event timestamps to keep the protocol correct: every completion is announced by the interactor (TDN/XDN) with its timestamp and duration, so a purely reactive scheduler needs no arithmetic on future times. If you simulate ahead for planning, use ordinary double-precision arithmetic with the piecewise-linear lookup and transfer formula above.
At most 2⋅106
frames per test; use fast I/O.
Degenerate values (e.g. 𝐾=1
, num_layers=1
) occur and simply disable the corresponding mechanic.
Output
After every frame (one turn), print a count 𝑛
(0≤𝑛≤𝐾+1
), followed by 𝑛
assignments. For readability the examples put one assignment on each line, but the parser accepts arbitrary whitespace between tokens. Each assignment has the form <server> <task_spec>. It starts one task on a computer that is currently free. The command words shown below are fixed output syntax; copy them exactly. Integer tokens use the ordinary signed-decimal syntax accepted by C++ conversion (an optional leading + and leading zeros are accepted); echoed integers are canonical decimal. Thus a response containing no assignments is simply

0
For a first scheduler, most responses may contain only zero or one task. Unmentioned resources stay idle. You may leave any free computer idle even when one or more legal tasks are available. Waiting can be a useful grouping choice, but do not create the stuck state with no future event described in Interaction. Every frame is a scheduling opportunity, including frames containing only transfers or arrivals. After reading the whole frame, you may use any predecessor or free resource reported anywhere in it.

When deciding whether a task is legal, ask three questions: Is its computer free? Has every required predecessor event arrived? Is every request in the task at exactly this step and not already in flight or finished? The tables below make these checks precise.

All assignments in one response begin simultaneously at timestamp 𝑡
. One cannot depend on another assignment from that same response. Assign at most one task to each resource; it becomes free again only when its TDN arrives. The six legal task shapes are:

step task_spec Server
input stage first step P PRE <remote> <rid> local computer (also fixes the remote computer)
input stage process P PROC <ls> <le> <remote> <rid> that remote computer
input stage final step P POST <remote> <rid> local computer
output first step D PRE -1 <m> <rid...> local computer (across remote computers)
output process D PROC <remote> <m> <rid...> that remote computer
output final step D POST -1 <m> <rid...> local computer (across remote computers)
For D PRE/D POST, the -1 marks a group spanning remote computers; for D PROC, all members must be assigned to <remote>. In P PRE, <remote> must be in [0,𝐾)
; in P PROC and P POST, <remote> must equal the request's assigned remote computer — anything else is a violation. (P POST itself runs on the local computer: its <remote> field carries no scheduling meaning and is purely a consistency check against the request's fixed assignment.) Iteration indices are never transmitted: each request's steps are strictly sequential, so your own bookkeeping is the sole — and sufficient — source of truth for which iteration a task belongs to.

The order of request ids within a group has no semantic effect: legality, durations, transfer sizes, and all subsequent events depend only on the member set (D PRE -1 3 4 7 9 and D PRE -1 3 9 4 7 denote the same task). Order is preserved on echo: TDN repeats the same fields in canonical decimal form with single spaces, and every XDN lists its request ids in the order of the triggering task's specification — a D PRE group spanning remote computers has per-remote computer local-to-remote transfer lists that remote computer's members as a subsequence of your D PRE line, an output-step remote-to-local transfer lists the triggering D PROC group in your order.

Dependency Summary

Each task is listed separately so its predecessor and completion effect are easy to scan. For a group, the predecessor rule applies to every member.

P PRE — Server: local computer. Requires: the request's ARR event. Completes: fixes the request's remote computer and queues its input-stage local-to-remote transfer.
P PROC piece — Server: the assigned remote computer. Requires: for the first piece, the input-stage local-to-remote XDN; for a later piece, the previous piece's TDN. Completes: only the last piece (𝑙𝑒=num_layers
) queues the input-stage remote-to-local transfer.
P POST — Server: local computer. Requires: the request's input-stage remote-to-local XDN. Completes: stops TDR and makes the request ready for output.
D PRE — Server: local computer. Requires: each member's previous final-step TDN (P POST or its previous D POST), and the request must not be finished. Completes: queues one local-to-remote transfer per remote computer represented in the group, in increasing remote-computer index.
D PROC — Server: the assigned remote computer. Requires: the local-to-remote XDN carrying each member's current-iteration data. Completes: queues one remote-to-local transfer whose length is the group size.
D POST — Server: local computer. Requires: the remote-to-local XDN carrying each member's current-iteration result. Completes: produces one token per member; after iteration 𝐿out[𝑖]
, produces FIN.
Interaction
Each test is a separate run of your program. Flush the output stream after every response. In C++, use cout « flush; after printing the complete response; we recommend ios::sync_with_stdio(false); cin.tie(nullptr); for fast input. In Python, call sys.stdout.flush() after printing the complete response (or use print(..., flush=True) for its final line).

The interaction proceeds: startup configuration (2 lines) →
Task-Time Table (𝑁+1
lines) →
repeat {
frame →
your response}

END. Your first response follows the first frame. The whole program follows this loop:

read the 2 parameter lines, then N and the N warmup rows
loop:
read one line; if it is END: exit
parse it as timestamp t; read event count e from the next line
read the e event lines
update your state (completions, arrivals, transfers, FINs)
choose assignments for currently free resources (possibly none)
print n and the n assignment lines; flush
For a simple implementation, store one state per request. Its path is

ARR -> P PRE -> input stage UP -> P PROC piece(s) -> input stage DOWN -> P POST
-> D PRE -> output UP -> D PROC -> output DOWN -> D POST
-> either FIN or the next D PRE
Move a request to its next state only when the corresponding event appears in a frame. Mark a computer busy when you assign it and free only when its TDN appears. Tracking these states is enough for a correct solution. You do not need to predict when tasks will finish.

You act only when a frame arrives. There is no timer or self-wake mechanism: you cannot ask to be woken at a chosen future time, and you cannot deliberately idle a computer until an arbitrary moment. A task can only be assigned in response to a frame, at that frame's timestamp. If you delay an action, you must wait for a future event frame.

Mistakes That Give Zero Points on a Test

The following errors score 0
on the test. A legal but slow choice is still valid; it only lowers your score.

Assigning a task to a busy computer, or two tasks to one resource in a single response.
Assigning a task before all of its required earlier events have been delivered. This includes referencing a rid that has not arrived and re-issuing a completed step.
Including a request that is already part of an in-flight task, or that has already finished (FIN).
Wrong remote computer: P PROC/P POST with a remote computer other than the request's assigned remote computer; P PRE with a remote computer outside [0,𝐾)
; a D PROC member assigned to a different remote computer.
Illegal piece: empty (𝑙𝑠=𝑙𝑒
), outside [0,num_layers]
, or not ascending and gap-free.
Malformed group: 𝑚<1
or duplicate request ids.
Malformed or unparsable output.
Reaching a stuck state with unfinished requests and no possible future event (see below).
Exceeding the time or memory limit.
Errors and stream closure

The interactor never produces -1 as a failure response. This is unrelated to the mandatory -1 marker in participant commands D PRE and D POST. On a protocol violation or malformed response it simply stops: the test scores 0
and no further frames are sent. If reading input ever fails or reaches end-of-file, exit immediately with exit code 0
— do not block waiting for more input and do not crash; the verdict is determined by the interactor, not by your exit path.

Getting Stuck

If unfinished requests remain but no task, transfer, or future arrival can create another event, the run is stuck. The interactor detects this stuck state, terminates, and assigns 0
to the test. While future arrivals remain, responding with 0
assignments is safe because time advances to the next arrival. Delaying a request hurts the score; permanently abandoning it can cause this stuck state.

Termination

When all requests have finished, the interactor sends END (a single line, in place of the next frame's header) after reading your response to the final frame. Read it and exit.

The total number of requests 𝑅
is not announced in advance and there is no "no more arrivals" signal — you cannot distinguish a lull from the end of the stream. This is intentional; plan your grouping accordingly.

Scoring
You do not need to calculate the score to write a valid scheduler. First make a scheduler that finishes every request legally. Read the formula only when you are ready to improve its score. It balances two intuitive goals:

Output rate: finish more output tokens per millisecond of simulated time.
waiting time: make requests ready for output promptly and avoid long gaps between produced tokens.
Each goal becomes a component in [0,1]
. The weights 𝑤tp
and 𝑤𝑐
say how much that test values each goal, and the final score is their weighted sum times 1000
.

The exact guarantees are: 𝑤tp,𝑤𝑐≥0
; 𝑤tp+𝑤𝑐=1
; SLO1>0
; SLO2>0
; 𝑡𝑝UB>𝑡𝑝base
; 𝑑𝑖𝑠𝑡base≥0
. Either weight may be 0
, but every request must still be completed and every input/output rule obeyed. The following formula turns a value into a number from 0
to 1
:

𝖼𝗅𝖺𝗆𝗉(𝑥;base,target)=max(0,min(1,(𝑥−base)/(target−base)))

(0
at base
, 1
at target
, clamped outside).

Output-Rate Component. tp=∑𝑖𝐿out[𝑖]/total elapsed time
tokens/ms, where total elapsed time =
(the latest final-token production time over all requests) −
(the earliest request arrival time), in ms. Output rate component =𝖼𝗅𝖺𝗆𝗉(tp;𝑡𝑝base,𝑡𝑝UB)
. The value 𝑡𝑝base
comes from a fixed one-request-at-a-time reference schedule; 𝑡𝑝UB
is an estimated high rate. At or below the baseline the component is 0
; at or above the upper bound it is 1
; between them it grows linearly. These values set scoring and are not promises about any solution in the testing kit.

Waiting-Time Component. The input gives two waiting-time targets. SLO1 is the target average wait until a request is ready for its first output step (time to decode ready, TDR). No token has been produced at this point. SLO2 is the target mean gap between consecutive produced tokens (TPOT). The letters SLO are only part of the fixed input names. For request 𝑖
, let 𝑒1<𝑒2<⋯<𝑒𝐿out[𝑖]
be its token production times, i.e. the completion times of its output D POST tasks.

tdr

mean over all requests of (input stage final step completion −
arrival).
tpot

mean of 𝑒𝑗+1−𝑒𝑗
over all consecutive output gaps of all requests pooled (1≤𝑗<𝐿out[𝑖]
). A request with 𝐿out[𝑖]=1
contributes no gaps. If no request contributes a gap, tpot
is defined as 0
.
excess_tdr=max(0,tdr−SLO1SLO1),excess_tpot=max(0,tpot−SLO2SLO2),
dist=excess_tdr2+excess_tpot2‾‾‾‾‾‾‾‾‾‾‾‾‾‾‾‾‾‾‾‾‾‾‾‾‾√.

waiting-time component =𝖼𝗅𝖺𝗆𝗉(dist;𝑑𝑖𝑠𝑡base,0)
— the same conversion with lower values treated as better: lower dist
is better, 1
at dist=0
, 0
at the reference scheduler's amount above the waiting-time targets 𝑑𝑖𝑠𝑡base
. Written out directly (the form easiest to implement and verify):

waiting-time component=⎧⎩⎨⎪⎪max(0,1−dist/𝑑𝑖𝑠𝑡base)10if 𝑑𝑖𝑠𝑡base>0,if 𝑑𝑖𝑠𝑡base=0 and dist=0,if 𝑑𝑖𝑠𝑡base=0 and dist>0.

Important special case: if 𝑑𝑖𝑠𝑡base=0
, the waiting-time component is either 0
or 1
. It is 1
only when both mean-waiting-time targets are met (dist=0
), and 0
otherwise.

Diagram guide: the upper branch rewards higher output rate; the lower branch rewards smaller normalized TDR/TPOT target violations. The two [0,1]
components are weighted, added, and multiplied by 1000
.

𝖭𝗈𝗋𝗆𝖺𝗅𝗂𝗓𝖾𝖽𝖲𝖼𝗈𝗋𝖾=𝑤tp⋅𝖼𝗅𝖺𝗆𝗉(tp;𝑡𝑝base,𝑡𝑝UB)+𝑤𝑐⋅𝖼𝗅𝖺𝗆𝗉(dist;𝑑𝑖𝑠𝑡base,0),
𝖲𝖼𝗈𝗋𝖾=1000⋅𝖭𝗈𝗋𝗆𝖺𝗅𝗂𝗓𝖾𝖽𝖲𝖼𝗈𝗋𝖾.

Thus each completed test awards a score in [0,1000]
. Verdicts. A completed interaction scores as above. A protocol error, malformed output, stuck state with no future event, or exceeding the time/memory limit scores 0
on that test; other tests are unaffected. No partial credit is awarded for an unfinished test.

Contest aggregation. The 22 preliminary tests provide feedback and do not contribute to the final ranking. The final score is the arithmetic mean of the 20 frozen final-test scores. Ranking uses that mean before display rounding; the displayed score is rounded to three digits after the decimal point. Therefore a known per-test vector of 0
, 500
, and 1000
-point outcomes contributes exactly those values before averaging. This problem is not open to hacks.

Example
InputCopy
1 1.000000000 2.000000000 1.000000000 125000 4
30.000000000 15.000000000 0.062500000 0.022222222 0.000000000 0.500000000 0.500000000
2
1 3.000000000 10.000000000 2.000000000 1.000000000 4.000000000 1.000000000
4 3.000000000 10.000000000 2.000000000 1.000000000 4.000000000 1.000000000
0.000000000
1
ARR 0 4
4.000000000
1
TDN E P PRE 0 0 3.000000000
10.000000000
1
XDN UP 0 500000 PRE 1 0
21.000000000
1
TDN C0 P PROC 0 4 0 0 10.000000000
27.000000000
1
XDN DOWN 0 500000 PRE 1 0
30.000000000
1
TDN E P POST 0 0 2.000000000
32.000000000
1
TDN E D PRE -1 1 0 1.000000000
35.000000000
1
XDN UP 0 125000 DEC 1 0
40.000000000
1
TDN C0 D PROC 0 1 0 4.000000000
43.000000000
1
XDN DOWN 0 125000 DEC 1 0
45.000000000
2
TDN E D POST -1 1 0 1.000000000
FIN 0
END
OutputCopy
1
E P PRE 0 0
0
1
C0 P PROC 0 4 0 0
0
1
E P POST 0 0
1
E D PRE -1 1 0
0
1
C0 D PROC 0 1 0
0
1
E D POST -1 1 0
0
Note
Examples

If interactive problems are new to you, begin with Example 1 and focus on the repeating pattern: read a complete frame on the left, update your state, then print one complete response on the right. The exact timestamps are consequences of the supplied costs; your scheduler never has to predict them. The transcript below shows the participant-side input and output directly; output responses contain no timestamps.

The examples use 𝑆=1
, latency_in_ms=2
, bandwidth_gbps=1
, and bytes_per_token=125000
. Because bandwidth is in gigabits per second, a one-token transfer takes 2+8⋅125000/106=3
ms. Every table has the interactor's input on the left and your complete response on the right. Each protocol line occupies its own table row. A response begins with its assignment count.

Example 1: one request, end to end.

Here 𝐾=1
, num_layers=4
, request 0
has 𝐿in[0]=4
and 𝐿out[0]=1
, and the relevant execution durations are 3,10,2
ms for input stage and 1,4,1
ms for output. The scheduler uses one full input-stage piece.

Interactor Your response
0.000000000 1
1 E P PRE 0 0
ARR 0 4
4.000000000 0
1
TDN E P PRE 0 0 3.000000000
10.000000000 1
1 C0 P PROC 0 4 0 0
XDN UP 0 500000 PRE 1 0
21.000000000 0
1
TDN C0 P PROC 0 4 0 0 10.000000000
27.000000000 1
1 E P POST 0 0
XDN DOWN 0 500000 PRE 1 0
30.000000000 1
1 E D PRE -1 1 0
TDN E P POST 0 0 2.000000000
32.000000000 0
1
TDN E D PRE -1 1 0 1.000000000
35.000000000 1
1 C0 D PROC 0 1 0
XDN UP 0 125000 DEC 1 0
40.000000000 0
1
TDN C0 D PROC 0 1 0 4.000000000
43.000000000 1
1 E D POST -1 1 0
XDN DOWN 0 125000 DEC 1 0
45.000000000 0
2
TDN E D POST -1 1 0 1.000000000
FIN 0
END exit
The input-stage final step completes at 30.000
, so this request's TDR is 30
ms. Its only token is produced at 45.000
. Because 𝐿out[0]=1
, it contributes no TPOT gap.

Example 2: grouping requests from two remote computers.

Assume requests 0
and 1
have already been assigned to C0 and C1, respectively, and both are ready for their first output step at time 16.500
. This excerpt shows one across-remote-computers first step group, two single-remote-computer process tasks, and one across-remote-computers final step group.

Interactor Your response
16.500000000 1
1 E D PRE -1 2 0 1
TDN E P POST 1 1 1.250000000
19.000000000 0
1
TDN E D PRE -1 2 0 1 1.500000000
22.000000000 1
1 C0 D PROC 0 1 0
XDN UP 0 125000 DEC 1 0
25.000000000 1
1 C1 D PROC 1 1 1
XDN UP 1 125000 DEC 1 1
27.000000000 0
1
TDN C0 D PROC 0 1 0 4.000000000
30.000000000 0
2
TDN C1 D PROC 1 1 1 4.000000000
XDN DOWN 0 125000 DEC 1 0
33.000000000 1
1 E D POST -1 2 0 1
XDN DOWN 1 125000 DEC 1 1
35.500000000 0
3
TDN E D POST -1 2 0 1 1.500000000
FIN 0
FIN 1
END exit
The two requests may share the local computer tasks even though their process tasks run on different remote computers. The two local-to-remote transfers are handled one at a time, in remote-computer number order; the final D POST combines results delivered by two different remote-to-local transfers.

An output task's time depends only on its group size, never on which output token is being generated or how many tokens are in the request's previously stored data. The task-time table and transfer formula are the complete cost model.

ICPC 2026 Online Challenge 1 powered by Huawei
Contest is running
2 weeks
Contestant
Add to favourites
→ Submit?
Language:
GNU G++23 14.2 (64 bit, msys2)
Choose file: No file chosen
→ Last submissions
Submission Time Verdict
387019200 Aug/14/2026 14:06 Partial result: 12912.426 points
→ Your points
Points
A 12912.426
Total 12912.426

Réflexion

Compiled round-robin cloud scheduler and evaluated program correctness

Worked for 5m 51s

Réponse

Your biggest loss is not cloud assignment; it is singleton decode. The judge data makes that especially clear: test #19 is effectively pure-throughput (w_tp=1) and your throughput component is only 0.001948, while #3/#7 are effectively pure-SLO tests. #22 is roughly 50/50 and has excellent SLO but almost zero throughput. So one fixed FIFO policy cannot score well across the set.

The replacement below makes four high-impact changes:

  • Batches D PRE, D PROC, and D POST using the supplied task-time curves instead of always using batch size 1.
  • Chooses a batch-size target from the actual scoring parameters (w_tp, w_c, tp_base, tp_UB, SLO2, dist_base).
  • Uses SLO-aware scheduling: requests approaching TDR/TPOT deadlines can override throughput ordering.
  • Splits long P PROC work when latency matters, but avoids splitting in throughput-heavy tests and avoids splitting when the request's own TDR is already tight.
  • Adds starvation protection so prefill still progresses under continuous decode load.
  • Balances requests across clouds with round-robin tie-breaking.

There is no scheduler that can guarantee the mathematical maximum because future arrivals and Lout are hidden, but this attacks the main structural weaknesses in your 12,912-point solution. I also compiled it with G++23 and checked that it produces the correct public-example interaction.

cpp
#include <bits/stdc++.h> using namespace std; enum class State { NONE, P_PRE_READY, P_PRE_RUN, WAIT_P_UP, P_PROC_READY, P_PROC_RUN, WAIT_P_DOWN, P_POST_READY, P_POST_RUN, D_PRE_READY, D_PRE_RUN, WAIT_D_UP, D_PROC_READY, D_PROC_RUN, WAIT_D_DOWN, D_POST_READY, D_POST_RUN, FINISHED }; struct Request { int lin = 0; int remote = -1; int nextLayer = 0; State st = State::NONE; bool finished = false; bool decodeStarted = false; int tokensDone = 0; double arrival = 0.0; double lastToken = -1.0; }; struct ReadyItem { int rid; State expected; }; struct Curve { vector<pair<int, double>> p; double get(int x) const { if (x <= p.front().first) return p.front().second; if (x >= p.back().first) return p.back().second; int lo = 0; int hi = (int)p.size() - 1; while (lo + 1 < hi) { int md = (lo + hi) >> 1; if (p[md].first <= x) lo = md; else hi = md; } auto [x0, y0] = p[lo]; auto [x1, y1] = p[hi]; double a = double(x - x0) / double(x1 - x0); return y0 + (y1 - y0) * a; } }; enum Col { PPRE = 0, PPROC = 1, PPOST = 2, DPRE = 3, DPROC = 4, DPOST = 5 }; enum class EdgeChoice { NONE, PPRE, PPOST, DPRE, DPOST }; enum class CloudChoice { NONE, PPROC, DPROC }; int main() { ios::sync_with_stdio(false); cin.tie(nullptr); // ------------------------------------------------------------ // Configuration // ------------------------------------------------------------ int K; double S; double latency_in_ms; double bandwidth_gbps; long long bytes_per_token; int num_layers; if (!(cin >> K >> S >> latency_in_ms >> bandwidth_gbps >> bytes_per_token >> num_layers)) { return 0; } double SLO1, SLO2; double tp_UB, tp_base, dist_base; double w_tp, w_c; cin >> SLO1 >> SLO2 >> tp_UB >> tp_base >> dist_base >> w_tp >> w_c; // ------------------------------------------------------------ // Task curves // ------------------------------------------------------------ int N; cin >> N; array<Curve, 6> curve; for (int i = 0; i < N; ++i) { int bs; double v[6]; cin >> bs; for (double& x : v) cin >> x; for (int c = 0; c < 6; ++c) { if (v[c] >= 0.0) curve[c].p.push_back({bs, v[c]}); } } for (auto& c : curve) sort(c.p.begin(), c.p.end()); // milliseconds per one token-sized item in the bandwidth term const double beta = 8.0 * double(bytes_per_token) / (bandwidth_gbps * 1e6); auto tx = [&](long long len) -> double { return latency_in_ms + beta * double(len); }; auto svc = [&](int col, int batch) -> double { return S + curve[col].get(batch); }; auto normTp = [&](double x) -> double { double z = (x - tp_base) / (tp_UB - tp_base); return max(0.0, min(1.0, z)); }; auto normWait = [&](double dist) -> double { if (dist_base > 0.0) return max(0.0, 1.0 - dist / dist_base); return dist <= 1e-15 ? 1.0 : 0.0; }; // ------------------------------------------------------------ // Find a useful steady-state decode batch size. // // We model a cohort of m requests distributed across up to K // clouds. Throughput is estimated from the bottleneck resource. // TPOT is estimated by conservative end-to-end cohort latency. // // Then use the ACTUAL scoring weights to choose m. // ------------------------------------------------------------ const int MAX_R = 2000; int decodeTarget = 1; double decodeTargetLatency = 0.0; double bestProxy = -1e100; double bestRawTp = -1.0; double bestLatency = 1e100; for (int m = 1; m <= MAX_R; ++m) { int usedClouds = min(K, m); int perCloud = (m + usedClouds - 1) / usedClouds; double preService = svc(DPRE, m); double postService = svc(DPOST, m); double procService = svc(DPROC, perCloud); // A D PRE spanning usedClouds clouds creates one UP // transfer for each cloud. double upLoad = usedClouds * latency_in_ms + beta * m; // Assuming balanced D PROC grouping, the same aggregate // structure occurs on DOWN. double downLoad = upLoad; // Steady-state throughput proxy. // // PRE+POST share the same edge, while UP and DOWN have // independent FIFO directions. double bottleneck = max({ preService + postService, procService, upLoad, downLoad }); double tpProxy = double(m) / bottleneck; // Conservative request-cycle latency estimate. double latencyProxy = preService + upLoad + procService + downLoad + postService; double excessTpot = max( 0.0, (latencyProxy - SLO2) / SLO2 ); // Decode doesn't affect TDR once P POST is done, so only // TPOT is used in this batch-size proxy. double waitingProxy = normWait(excessTpot); double scoreProxy = w_tp * normTp(tpProxy) + w_c * waitingProxy; bool better = false; if (scoreProxy > bestProxy + 1e-12) { better = true; } else if ( abs(scoreProxy - bestProxy) <= 1e-12 && tpProxy > bestRawTp + 1e-12 ) { better = true; } else if ( abs(scoreProxy - bestProxy) <= 1e-12 && abs(tpProxy - bestRawTp) <= 1e-12 && latencyProxy < bestLatency - 1e-12 ) { better = true; } if (better) { bestProxy = scoreProxy; bestRawTp = tpProxy; bestLatency = latencyProxy; decodeTarget = m; decodeTargetLatency = latencyProxy; } } // Approximate local batch size on each cloud if the global // cohort is balanced. int cloudDecodeTarget = max( 1, (decodeTarget + K - 1) / K ); // ------------------------------------------------------------ // Dynamic state // ------------------------------------------------------------ vector<Request> req; bool edgeBusy = false; vector<bool> cloudBusy(K, false); // Requests assigned to each cloud but not FINished. vector<int> activeOnCloud(K, 0); // Requests that have completed P POST and are in decode. // Used to determine whether a long P PROC should be chunked. vector<int> decodeLive(K, 0); // Starvation protection. int edgeDecodeBurst = 0; vector<int> cloudDecodeBurst(K, 0); int rrCloud = 0; deque<ReadyItem> qPPre; deque<ReadyItem> qPPost; deque<ReadyItem> qDPre; deque<ReadyItem> qDPost; vector<deque<ReadyItem>> qPProc(K); vector<deque<ReadyItem>> qDProc(K); auto ensureRequest = [&](int rid) { if ((int)req.size() <= rid) req.resize(rid + 1); }; auto push = [&](deque<ReadyItem>& q, int rid, State st) { q.push_back({rid, st}); }; // ------------------------------------------------------------ // Lazy-ready-queue helpers // ------------------------------------------------------------ auto valid = [&](const ReadyItem& x, State st, int cloud) -> bool { if (x.rid < 0 || x.rid >= (int)req.size()) return false; const Request& r = req[x.rid]; if (r.finished || r.st != st || x.expected != st) return false; if (cloud >= 0 && r.remote != cloud) return false; return true; }; auto cleanFront = [&](deque<ReadyItem>& q, State st, int cloud) { while (!q.empty() && !valid(q.front(), st, cloud)) { q.pop_front(); } }; auto peekRid = [&](deque<ReadyItem>& q, State st, int cloud) -> int { cleanFront(q, st, cloud); if (q.empty()) return -1; return q.front().rid; }; auto takeGroup = [&](deque<ReadyItem>& q, State st, int cloud, int limit) { vector<int> result; result.reserve(limit); while ((int)result.size() < limit) { cleanFront(q, st, cloud); if (q.empty()) break; int rid = q.front().rid; q.pop_front(); if (rid < 0 || rid >= (int)req.size()) continue; const Request& r = req[rid]; if (r.finished || r.st != st) continue; if (cloud >= 0 && r.remote != cloud) continue; result.push_back(rid); } return result; }; // ------------------------------------------------------------ // Cloud selection // // First balance live request count. Round-robin breaks ties // so synchronized cohorts naturally spread across clouds. // ------------------------------------------------------------ auto chooseCloud = [&]() -> int { int best = -1; for (int d = 0; d < K; ++d) { int k = (rrCloud + d) % K; if (best == -1) { best = k; continue; } if (activeOnCloud[k] != activeOnCloud[best]) { if (activeOnCloud[k] < activeOnCloud[best]) { best = k; } continue; } int qk = (int)qPProc[k].size() + (int)qDProc[k].size(); int qb = (int)qPProc[best].size() + (int)qDProc[best].size(); if (qk != qb) { if (qk < qb) best = k; continue; } if (cloudBusy[k] != cloudBusy[best] && !cloudBusy[k]) { best = k; } } rrCloud = (best + 1) % K; return best; }; // ------------------------------------------------------------ // Approximate remaining times used for SLO-aware EDF. // ------------------------------------------------------------ auto ppreRemaining = [&](const Request& r) -> double { return svc(PPRE, r.lin) + tx(r.lin) + svc(PPROC, r.lin) + tx(r.lin) + svc(PPOST, r.lin); }; auto pprocRemaining = [&](const Request& r) -> double { double frac = double(num_layers - r.nextLayer) / double(num_layers); return S + curve[PPROC].get(r.lin) * frac + tx(r.lin) + svc(PPOST, r.lin); }; auto ppostRemaining = [&](const Request& r) -> double { return svc(PPOST, r.lin); }; auto dpreRemaining = [&](const Request&) -> double { return decodeTargetLatency; }; auto dprocRemaining = [&](const Request&) -> double { return svc(DPROC, cloudDecodeTarget) + tx(cloudDecodeTarget) + svc(DPOST, decodeTarget); }; auto dpostRemaining = [&](const Request&) -> double { return svc(DPOST, decodeTarget); }; // Normalized deadline slack. // // Negative => predicted SLO miss unless accelerated. // // First-token decode has no TPOT deadline, so it deliberately // receives no TPOT urgency. auto normSlack = [&](int rid, State st, double now) -> double { if (rid < 0) return 1e100; const Request& r = req[rid]; if (st == State::P_PRE_READY) { return ( r.arrival + SLO1 - now - ppreRemaining(r) ) / SLO1; } if (st == State::P_PROC_READY) { return ( r.arrival + SLO1 - now - pprocRemaining(r) ) / SLO1; } if (st == State::P_POST_READY) { return ( r.arrival + SLO1 - now - ppostRemaining(r) ) / SLO1; } if (r.tokensDone <= 0 || r.lastToken < 0.0) { return 1e100; } if (st == State::D_PRE_READY) { return ( r.lastToken + SLO2 - now - dpreRemaining(r) ) / SLO2; } if (st == State::D_PROC_READY) { return ( r.lastToken + SLO2 - now - dprocRemaining(r) ) / SLO2; } if (st == State::D_POST_READY) { return ( r.lastToken + SLO2 - now - dpostRemaining(r) ) / SLO2; } return 1e100; }; // In throughput-heavy cases EDF only overrides after an // expected miss. In latency-heavy cases it acts earlier. auto urgencyShouldOverride = [&](double minSlack) -> bool { if (w_c <= 1e-12 || minSlack > 1e90) return false; if (w_c >= 0.95) return minSlack < 0.75; if (w_c >= 0.80) return minSlack < 0.40; if (w_c >= 0.55) return minSlack < 0.10; return minSlack < 0.0; }; // ------------------------------------------------------------ // Prefill chunking // // Only split if: // * latency has meaningful score weight; // * existing decode traffic uses this cloud; // * TDR has enough slack to pay the extra S costs. // ------------------------------------------------------------ auto chooseChunkEnd = [&](int k, const Request& r, double now) -> int { int remaining = num_layers - r.nextLayer; if (remaining <= 1) return num_layers; // Throughput-only tests should usually pay S once. if (w_c <= 0.05 || decodeLive[k] <= 0) { return num_layers; } double fullDur = curve[PPROC].get(r.lin); double remDur = fullDur * double(remaining) / double(num_layers); // Do not sacrifice this request's TDR when it is // already close to / beyond the target. double optimisticFinish = now + S + remDur + tx(r.lin) + svc(PPOST, r.lin); double tdrSlack = r.arrival + SLO1 - optimisticFinish; if (tdrSlack <= 0.5 * S) return num_layers; // Try not to monopolize a cloud for much more than a // fraction of the TPOT target. double minLayerHold = S + fullDur / double(num_layers); double quantum = max( minLayerHold, 0.35 * SLO2 ); int idealPieces = max( 1, (int)ceil( (S + remDur) / max(quantum, 1e-12) ) ); idealPieces = min( idealPieces, remaining ); // Blend continuously according to waiting-score weight. int weightedPieces = 1 + (int)llround( w_c * double(idealPieces - 1) ); // Each additional piece pays an additional S. int maxByTdr = 1 + (int)floor( max(0.0, tdrSlack) / S ); int pieces = max( 1, min({ weightedPieces, maxByTdr, remaining }) ); if (pieces <= 1) return num_layers; int chunkLayers = (remaining + pieces - 1) / pieces; return min( num_layers, r.nextLayer + max(1, chunkLayers) ); }; // ------------------------------------------------------------ // Interaction // ------------------------------------------------------------ while (true) { string first; if (!(cin >> first)) return 0; if (first == "END") return 0; double now = stod(first); int eventCount; cin >> eventCount; // -------------------------------------------------------- // Read the complete frame first. // -------------------------------------------------------- for (int ev = 0; ev < eventCount; ++ev) { string type; cin >> type; // ==================================================== // ARR // ==================================================== if (type == "ARR") { int rid, lin; cin >> rid >> lin; ensureRequest(rid); Request& r = req[rid]; r = Request{}; r.lin = lin; r.arrival = now; r.st = State::P_PRE_READY; push( qPPre, rid, State::P_PRE_READY ); } // ==================================================== // TDN // ==================================================== else if (type == "TDN") { string server; string family; string step; cin >> server >> family >> step; if (server == "E") { edgeBusy = false; } else { int k = stoi(server.substr(1)); cloudBusy[k] = false; } // ---------------- P ---------------- if (family == "P") { if (step == "PRE") { int remote, rid; double dur; cin >> remote >> rid >> dur; ensureRequest(rid); if (!req[rid].finished) { req[rid].st = State::WAIT_P_UP; } } else if (step == "PROC") { int ls, le; int remote, rid; double dur; cin >> ls >> le >> remote >> rid >> dur; ensureRequest(rid); Request& r = req[rid]; if (!r.finished) { r.nextLayer = le; if (le < num_layers) { r.st = State::P_PROC_READY; push( qPProc[remote], rid, State::P_PROC_READY ); } else { r.st = State::WAIT_P_DOWN; } } } else { // POST int remote, rid; double dur; cin >> remote >> rid >> dur; ensureRequest(rid); Request& r = req[rid]; if (!r.finished) { if (!r.decodeStarted) { r.decodeStarted = true; ++decodeLive[r.remote]; } r.st = State::D_PRE_READY; push( qDPre, rid, State::D_PRE_READY ); } } } // ---------------- D ---------------- else { if (step == "PRE") { int marker, m; cin >> marker >> m; vector<int> ids(m); for (int& rid : ids) cin >> rid; double dur; cin >> dur; for (int rid : ids) { ensureRequest(rid); if (!req[rid].finished) { req[rid].st = State::WAIT_D_UP; } } } else if (step == "PROC") { int remote, m; cin >> remote >> m; vector<int> ids(m); for (int& rid : ids) cin >> rid; double dur; cin >> dur; for (int rid : ids) { ensureRequest(rid); if (!req[rid].finished) { req[rid].st = State::WAIT_D_DOWN; } } } else { // POST int marker, m; cin >> marker >> m; vector<int> ids(m); for (int& rid : ids) cin >> rid; double dur; cin >> dur; for (int rid : ids) { ensureRequest(rid); Request& r = req[rid]; /* * If this is the final token, FIN in this * frame will invalidate this queue entry. */ if (!r.finished) { ++r.tokensDone; r.lastToken = now; r.st = State::D_PRE_READY; push( qDPre, rid, State::D_PRE_READY ); } } } } } // ==================================================== // XDN // ==================================================== else if (type == "XDN") { string direction; int remote; long long sizeBytes; string phase; int m; cin >> direction >> remote >> sizeBytes >> phase >> m; vector<int> ids(m); for (int& rid : ids) cin >> rid; if (phase == "PRE") { for (int rid : ids) { ensureRequest(rid); Request& r = req[rid]; if (r.finished) continue; if (direction == "UP") { r.st = State::P_PROC_READY; push( qPProc[remote], rid, State::P_PROC_READY ); } else { r.st = State::P_POST_READY; push( qPPost, rid, State::P_POST_READY ); } } } else { // DEC for (int rid : ids) { ensureRequest(rid); Request& r = req[rid]; if (r.finished) continue; if (direction == "UP") { r.st = State::D_PROC_READY; push( qDProc[remote], rid, State::D_PROC_READY ); } else { r.st = State::D_POST_READY; push( qDPost, rid, State::D_POST_READY ); } } } } // ==================================================== // FIN // ==================================================== else if (type == "FIN") { int rid; cin >> rid; ensureRequest(rid); Request& r = req[rid]; if (!r.finished) { r.finished = true; r.st = State::FINISHED; if (0 <= r.remote && r.remote < K) { --activeOnCloud[r.remote]; if (r.decodeStarted) --decodeLive[r.remote]; } } } } // -------------------------------------------------------- // Schedule. // -------------------------------------------------------- vector<string> answer; answer.reserve(K + 1); // ======================================================== // EDGE // ======================================================== if (!edgeBusy) { int rpPre = peekRid( qPPre, State::P_PRE_READY, -1 ); int rpPost = peekRid( qPPost, State::P_POST_READY, -1 ); int rdPre = peekRid( qDPre, State::D_PRE_READY, -1 ); int rdPost = peekRid( qDPost, State::D_POST_READY, -1 ); EdgeChoice choice = EdgeChoice::NONE; // First find the most SLO-urgent available task. double bestSlack = 1e100; auto consider = [&](int rid, State st, EdgeChoice c) { if (rid < 0) return; double sl = normSlack( rid, st, now ); if (sl < bestSlack) { bestSlack = sl; choice = c; } }; consider( rpPre, State::P_PRE_READY, EdgeChoice::PPRE ); consider( rpPost, State::P_POST_READY, EdgeChoice::PPOST ); consider( rdPre, State::D_PRE_READY, EdgeChoice::DPRE ); consider( rdPost, State::D_POST_READY, EdgeChoice::DPOST ); // If no SLO emergency, use throughput-oriented order. if (!urgencyShouldOverride(bestSlack)) { choice = EdgeChoice::NONE; bool pAvailable = rpPre >= 0 || rpPost >= 0; // Never allow decode to permanently starve prefill. if (pAvailable && edgeDecodeBurst >= 4) { choice = rpPost >= 0 ? EdgeChoice::PPOST : EdgeChoice::PPRE; } else if (w_c < 0.10) { // Pure/near-pure throughput. if (rdPost >= 0) choice = EdgeChoice::DPOST; else if (rdPre >= 0) choice = EdgeChoice::DPRE; else if (rpPost >= 0) choice = EdgeChoice::PPOST; else if (rpPre >= 0) choice = EdgeChoice::PPRE; } else { // Mixed objective: P POST is valuable because // it stops TDR immediately. if (rdPost >= 0) choice = EdgeChoice::DPOST; else if (rpPost >= 0) choice = EdgeChoice::PPOST; else if (rdPre >= 0) choice = EdgeChoice::DPRE; else if (rpPre >= 0) choice = EdgeChoice::PPRE; } } // ---------------- P PRE ---------------- if (choice == EdgeChoice::PPRE) { auto g = takeGroup( qPPre, State::P_PRE_READY, -1, 1 ); if (!g.empty()) { int rid = g[0]; int k = chooseCloud(); Request& r = req[rid]; r.remote = k; r.st = State::P_PRE_RUN; ++activeOnCloud[k]; answer.push_back( "E P PRE " + to_string(k) + " " + to_string(rid) ); edgeBusy = true; edgeDecodeBurst = 0; } } // ---------------- P POST ---------------- else if (choice == EdgeChoice::PPOST) { auto g = takeGroup( qPPost, State::P_POST_READY, -1, 1 ); if (!g.empty()) { int rid = g[0]; Request& r = req[rid]; r.st = State::P_POST_RUN; answer.push_back( "E P POST " + to_string(r.remote) + " " + to_string(rid) ); edgeBusy = true; edgeDecodeBurst = 0; } } // ---------------- D PRE ---------------- else if (choice == EdgeChoice::DPRE) { auto g = takeGroup( qDPre, State::D_PRE_READY, -1, decodeTarget ); if (!g.empty()) { string line = "E D PRE -1 " + to_string(g.size()); for (int rid : g) { req[rid].st = State::D_PRE_RUN; line += " " + to_string(rid); } answer.push_back(line); edgeBusy = true; ++edgeDecodeBurst; } } // ---------------- D POST ---------------- else if (choice == EdgeChoice::DPOST) { auto g = takeGroup( qDPost, State::D_POST_READY, -1, decodeTarget ); if (!g.empty()) { string line = "E D POST -1 " + to_string(g.size()); for (int rid : g) { req[rid].st = State::D_POST_RUN; line += " " + to_string(rid); } answer.push_back(line); edgeBusy = true; ++edgeDecodeBurst; } } } // ======================================================== // CLOUDS // ======================================================== for (int k = 0; k < K; ++k) { if (cloudBusy[k]) continue; int rp = peekRid( qPProc[k], State::P_PROC_READY, k ); int rd = peekRid( qDProc[k], State::D_PROC_READY, k ); CloudChoice choice = CloudChoice::NONE; double pSlack = normSlack( rp, State::P_PROC_READY, now ); double dSlack = normSlack( rd, State::D_PROC_READY, now ); double bestSlack = min(pSlack, dSlack); // SLO emergency. if (urgencyShouldOverride(bestSlack)) { if (rp >= 0 && pSlack <= dSlack) { choice = CloudChoice::PPROC; } else if (rd >= 0) { choice = CloudChoice::DPROC; } } // Normal operation. else { // Don't permanently starve input preparation. if (rp >= 0 && cloudDecodeBurst[k] >= 4) { choice = CloudChoice::PPROC; } else if (rd >= 0) { choice = CloudChoice::DPROC; } else if (rp >= 0) { choice = CloudChoice::PPROC; } } // ---------------- D PROC ---------------- if (choice == CloudChoice::DPROC) { auto g = takeGroup( qDProc[k], State::D_PROC_READY, k, cloudDecodeTarget ); if (!g.empty()) { string line = "C" + to_string(k) + " D PROC " + to_string(k) + " " + to_string(g.size()); for (int rid : g) { req[rid].st = State::D_PROC_RUN; line += " " + to_string(rid); } answer.push_back(line); cloudBusy[k] = true; ++cloudDecodeBurst[k]; } } // ---------------- P PROC ---------------- else if (choice == CloudChoice::PPROC) { auto g = takeGroup( qPProc[k], State::P_PROC_READY, k, 1 ); if (!g.empty()) { int rid = g[0]; Request& r = req[rid]; int ls = r.nextLayer; int le = chooseChunkEnd( k, r, now ); le = max( ls + 1, min(num_layers, le) ); r.st = State::P_PROC_RUN; answer.push_back( "C" + to_string(k) + " P PROC " + to_string(ls) + " " + to_string(le) + " " + to_string(k) + " " + to_string(rid) ); cloudBusy[k] = true; cloudDecodeBurst[k] = 0; } } } // -------------------------------------------------------- // One complete response. // -------------------------------------------------------- cout << answer.size() << '\n'; for (const string& line : answer) cout << line << '\n'; cout << flush; } }

What I would tune after the next submission

The next judge vector will be much more informative than the current one. In particular, watch #3, #7, #19, and #22:

  • #19 should jump substantially if batching is doing its job. If norm_tp is still poor, make throughput mode even more aggressive: remove the decode-burst limit when w_tp≈1 and consider batching all currently ready decode requests rather than capping at decodeTarget.
  • #3/#7 tell us whether the 0.35 × SLO2 P-PROC quantum is appropriate. If TPOT remains bad, reduce 0.35 toward 0.2; if TDR worsens, raise it toward 0.5.
  • #22 is the most useful mixed-objective test. Your current result already has norm_c≈0.992, so almost all remaining opportunity there is throughput. If the new scheduler sacrifices much of that 0.992, batching/chunking is too aggressive for mixed weights.

The single most important thing is to resubmit this version and give me the new 22-line judge output. With the before/after component values, the batch model and the three scheduling thresholds can be tuned specifically against the preliminary distribution rather than guessing.

Partager cette Q&R