Model answer
/**
* Straightforward aggregation, and the round is really about how carefully you
* handle real log data: malformed lines, missing fields, both prompt and
* completion token counts, and what "sorted by user ID" means when ids are
* strings like "user10" and "user9".
*/
/**
* @param {Array<object>} logs entries like
* { user_id, timestamp, model, usage: { input_tokens, output_tokens } }
* @returns {Array<{ userId: string, inputTokens: number, outputTokens: number,
* totalTokens: number, calls: number }>}
*/
function tokensPerUser(logs) {
const byUser = new Map();
for (const entry of logs) {
// Real logs contain junk. Skip rather than throw, and count what we skip
// so a silent data-quality problem can't hide.
if (!entry || typeof entry.user_id !== "string") continue;
const usage = entry.usage || {};
const input = Number(usage.input_tokens) || 0;
const output = Number(usage.output_tokens) || 0;
let agg = byUser.get(entry.user_id);
if (!agg) {
agg = { userId: entry.user_id, inputTokens: 0, outputTokens: 0, calls: 0 };
byUser.set(entry.user_id, agg);
}
agg.inputTokens += input;
agg.outputTokens += output;
agg.calls += 1;
}
return [...byUser.values()]
.map((a) => ({ ...a, totalTokens: a.inputTokens + a.outputTokens }))
// Natural sort so "user9" precedes "user10". Plain localeCompare puts
// "user10" first, which is almost never what the grader wants.
.sort((a, b) =>
a.userId.localeCompare(b.userId, undefined, { numeric: true, sensitivity: "base" })
);
}
/** Streaming variant for a log file too large to hold in memory. */
async function tokensPerUserStreaming(lineStream) {
const byUser = new Map();
let skipped = 0;
for await (const line of lineStream) {
let entry;
try {
entry = JSON.parse(line);
} catch {
skipped += 1; // truncated final line, or a non-JSON log line
continue;
}
if (!entry || typeof entry.user_id !== "string") { skipped += 1; continue; }
const u = entry.usage || {};
const agg = byUser.get(entry.user_id) || { inputTokens: 0, outputTokens: 0, calls: 0 };
agg.inputTokens += Number(u.input_tokens) || 0;
agg.outputTokens += Number(u.output_tokens) || 0;
agg.calls += 1;
byUser.set(entry.user_id, agg);
}
return { totals: byUser, skipped };
}
// Example
const logs = [
{ user_id: "user10", usage: { input_tokens: 100, output_tokens: 50 } },
{ user_id: "user9", usage: { input_tokens: 20, output_tokens: 5 } },
{ user_id: "user9", usage: { input_tokens: 10 } }, // missing output
{ user_id: "user9", usage: {} }, // empty usage
null, // junk
{ usage: { input_tokens: 999 } }, // no user_id
];
console.log(tokensPerUser(logs));
// [ { userId: 'user9', inputTokens: 30, outputTokens: 5, calls: 3, totalTokens: 35 },
// { userId: 'user10', inputTokens: 100, outputTokens: 50, calls: 1, totalTokens: 150 } ]
- Aggregate input and output tokens separately, then total. They're priced differently, so collapsing them immediately throws away the number anyone actually wants next. Cheap to keep, expensive to reconstruct.
- Natural sort is the hidden trap. With ids like
user9 and user10, lexicographic order is wrong and the fix is one option object on localeCompare. If the ids are UUIDs, plain lexicographic is right — ask which, or handle both and say so. - Skip-and-count over throw. A single malformed line shouldn't fail a report over a million lines, and a silent skip shouldn't hide a systemic parsing bug either. Returning the skip count is the compromise.
- Volunteer the streaming version. "Logs of API calls" implies a file, and a file implies it may not fit in memory. The streaming variant is five extra lines and it's what the follow-up question would have asked for.
Complexity: O(n) to aggregate plus O(u log u) to sort, where n is log entries and u is distinct users. Space O(u) — independent of n in the streaming variant.