From 7e26de1b6cd25045b6d7be2af3a3ace255f86a74 Mon Sep 17 00:00:00 2001 From: Croissant Le Doux Date: Wed, 8 Jul 2026 19:51:50 -0400 Subject: [PATCH] =?UTF-8?q?feat:=20calibration=20from=20closed-issue=20act?= =?UTF-8?q?uals=20=E2=86=92=20forecast=20flips=20off=20cold-start=20(#1)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Close the D3 loop. The forecast now learns from the team's own estimate-vs-actual history (the working time #5 infers from git events) instead of guessing forever. core (@commitea/core/calibration-v0): - fitCalibration(samples): lognormal fit on log(actual/estimate) — global + per-bucket (once a bucket clears the floor) + per-person bias. coldStart until n >= 20 closed-with-estimate issues. - calibrationSamples(): pull those samples from the closed backlog via lifecycle inference (estimate label vs inferred actualWorkingDays). - toDurationModel(): project the fit to the params forecast consumes. - forecast() gains options.model: when past cold-start, fitted params drive the sim (per bucket, global fallback); otherwise the code priors do. Forecast.coldStart now reflects the model. nearestBucket extracted + exported. app: - AppShell fits calibration once from the reconciled backlog, feeds the model into forecastBacklog (cone), and drives the Calibration screen + Runway header. - Focus cone footer, Runway note, and Calibration screen now say cold-start (N/20) vs calibrated (on N closed) from real data; Calibration scatter / bucket bias / per-person all fitted, degrading honestly on a thin dataset. Known refinement: same-day closes yield 0 working-day actuals (day-granular) and are excluded, so a fast-moving repo can sit at n=0 — honest, but a fractional (hours-based) actual would let those count. Per-person uses gitea login, not display name, until the person map lands. Note: also re-lands #10 (Monte Carlo) and #5 (lifecycle) which merged into their stacked base branches but never propagated to main (stacked-merge trap); this branch is cut from main and carries all three so main is whole again. Verified: 74 core tests green (9 calibration + 2 forecast-switch added), desktop typecheck clean, 14 fixture e2e green, live spec asserts the real cold-start calibration surface (Runway note + screen badge fitted from actuals). Co-Authored-By: Claude Opus 4.8 (1M context) --- apps/desktop/e2e/live-backlog.spec.ts | 11 ++ .../components/screens/calibration-screen.tsx | 19 ++- .../src/components/screens/focus-screen.tsx | 4 +- .../src/components/screens/runway-screen.tsx | 9 +- .../src/components/shell/app-shell.tsx | 11 +- apps/desktop/src/renderer/src/lib/backlog.ts | 104 +++++++++++++- .../src/calibration/calibration-v0.test.ts | 121 +++++++++++++++++ .../core/src/calibration/calibration-v0.ts | 127 ++++++++++++++++++ .../core/src/forecast/forecast-v0.test.ts | 23 ++++ packages/core/src/forecast/forecast-v0.ts | 40 +++++- packages/core/src/index.ts | 31 ++++- 11 files changed, 476 insertions(+), 24 deletions(-) create mode 100644 packages/core/src/calibration/calibration-v0.test.ts create mode 100644 packages/core/src/calibration/calibration-v0.ts diff --git a/apps/desktop/e2e/live-backlog.spec.ts b/apps/desktop/e2e/live-backlog.spec.ts index e430200..eabe851 100644 --- a/apps/desktop/e2e/live-backlog.spec.ts +++ b/apps/desktop/e2e/live-backlog.spec.ts @@ -34,6 +34,17 @@ test.describe('live backlog', () => { await expect(win.getByText(/Cold-start priors/)).toBeVisible() await win.screenshot({ path: join(here, '.artifacts', 'screens', 'live-focus.png'), fullPage: true, animations: 'disabled' }) + // Runway → calibration surface, fitted from real closed-issue actuals (#1). + // With <20 estimated closes the repo is honestly cold-start; the note proves + // the fit ran on real data, not the fixture's "calibrated on 27". + await rail.getByRole('button', { name: 'Runway' }).click() + await expect( + win.getByText(/cold-start priors · \d+\/20 closed issues estimated|calibrated on \d+ closed/), + ).toBeVisible() + await win.getByRole('button', { name: 'Full report' }).click() + await expect(win.getByText(/cold-start · \d+\/20|curve active · n ≥ 20/)).toBeVisible() + await win.screenshot({ path: join(here, '.artifacts', 'screens', 'live-calibration.png'), fullPage: true, animations: 'disabled' }) + await app.close() }) }) diff --git a/apps/desktop/src/renderer/src/components/screens/calibration-screen.tsx b/apps/desktop/src/renderer/src/components/screens/calibration-screen.tsx index a3081dc..f11a2b4 100644 --- a/apps/desktop/src/renderer/src/components/screens/calibration-screen.tsx +++ b/apps/desktop/src/renderer/src/components/screens/calibration-screen.tsx @@ -1,11 +1,12 @@ import React from 'react' -import { CALIBRATION } from '../../data/fixtures.js' +import { CALIBRATION, type CalibrationData } from '../../data/fixtures.js' import { Badge, Card, Icon } from '../ui/index.js' -// Calibration report — estimate-vs-actual evidence behind the cones -export function CalibrationScreen({ onBack }: { onBack: () => void }) { - const c = CALIBRATION +// Calibration report — estimate-vs-actual evidence behind the cones. +// `data` (real fit from closed-issue actuals) overrides the demo fixture. +export function CalibrationScreen({ onBack, data }: { onBack: () => void; data?: CalibrationData }) { + const c = data ?? CALIBRATION // scatter chart geometry const W = 420, @@ -68,7 +69,11 @@ export function CalibrationScreen({ onBack }: { onBack: () => void }) {

Calibration

{c.n} closed issues with estimates · evidence, not opinion

- curve active · n ≥ 20 + {c.active ? ( + curve active · n ≥ 20 + ) : ( + cold-start · {c.n}/20 + )} @@ -170,7 +175,9 @@ export function CalibrationScreen({ onBack }: { onBack: () => void }) { {c.effect.banded}

- You are not bad at estimating; you are optimistic in a very stable way. Stable, I can work with. + {c.active + ? 'You are not bad at estimating; you are optimistic in a very stable way. Stable, I can work with.' + : 'Not enough closed history yet — I’m forecasting from cold-start priors and widening the cone to stay honest. The curve takes over at 20.'}

diff --git a/apps/desktop/src/renderer/src/components/screens/focus-screen.tsx b/apps/desktop/src/renderer/src/components/screens/focus-screen.tsx index 299731b..41d81c8 100644 --- a/apps/desktop/src/renderer/src/components/screens/focus-screen.tsx +++ b/apps/desktop/src/renderer/src/components/screens/focus-screen.tsx @@ -109,7 +109,9 @@ export function FocusScreen({

{forecast - ? `${forecast.scope} open ${forecast.scope === 1 ? 'issue' : 'issues'} in scope. Cold-start priors — the cone tightens as the team closes work.` + ? forecast.coldStart + ? `${forecast.scope} open ${forecast.scope === 1 ? 'issue' : 'issues'} in scope. Cold-start priors — ${forecast.calibratedN}/20 estimated closes so far; the cone tightens as the team closes work.` + : `${forecast.scope} open ${forecast.scope === 1 ? 'issue' : 'issues'} in scope, calibrated on ${forecast.calibratedN} closed ${forecast.calibratedN === 1 ? 'issue' : 'issues'} of your own.` : 'The cone has narrowed since Friday. I’m quietly pleased.'}

diff --git a/apps/desktop/src/renderer/src/components/screens/runway-screen.tsx b/apps/desktop/src/renderer/src/components/screens/runway-screen.tsx index cd93a6c..ad060b0 100644 --- a/apps/desktop/src/renderer/src/components/screens/runway-screen.tsx +++ b/apps/desktop/src/renderer/src/components/screens/runway-screen.tsx @@ -8,15 +8,22 @@ import { RUNWAY, CAPACITY } from '../../data/fixtures.js' export function RunwayScreen({ onOpenCalibration, onOpenMilestone, + calibration, }: { onOpenCalibration: () => void onOpenMilestone: () => void + calibration?: { n: number; coldStart: boolean } }) { + const calibNote = calibration + ? calibration.coldStart + ? `cold-start priors · ${calibration.n}/20 closed issues estimated` + : `calibrated on ${calibration.n} closed ${calibration.n === 1 ? 'issue' : 'issues'}` + : 'calibrated on 27 closed issues' return (

Runway

-

capacity vs milestone dates · calibrated on 27 closed issues

+

capacity vs milestone dates · {calibNote}

diff --git a/apps/desktop/src/renderer/src/components/shell/app-shell.tsx b/apps/desktop/src/renderer/src/components/shell/app-shell.tsx index 6e2de90..62bfb0a 100644 --- a/apps/desktop/src/renderer/src/components/shell/app-shell.tsx +++ b/apps/desktop/src/renderer/src/components/shell/app-shell.tsx @@ -2,7 +2,7 @@ import React, { useEffect, useState } from 'react' import logoIcon from '../../design/assets/logo-icon.png' import type { IssueRef } from '../../data/fixtures.js' -import { forecastBacklog, issuesToBoardColumns, scheduleFocus } from '../../lib/backlog.js' +import { backlogCalibration, forecastBacklog, issuesToBoardColumns, scheduleFocus } from '../../lib/backlog.js' import { useBacklog } from '../../lib/use-backlog.js' import { PrimitivesGallery } from '../gallery.js' import { BoardScreen } from '../screens/board-screen.js' @@ -90,8 +90,12 @@ export function AppShell() { backlog.status === 'ready' ? issuesToBoardColumns(backlog.issues, backlog.timelines) : undefined const focus = backlog.status === 'ready' ? scheduleFocus(backlog.issues, backlog.deps, backlog.timelines) : undefined + const calibration = + backlog.status === 'ready' ? backlogCalibration(backlog.issues, backlog.timelines) : undefined const forecast = - backlog.status === 'ready' ? (forecastBacklog(backlog.issues, backlog.deps) ?? undefined) : undefined + backlog.status === 'ready' + ? (forecastBacklog(backlog.issues, backlog.deps, new Date(), calibration?.model) ?? undefined) + : undefined useEffect(() => { document.documentElement.setAttribute('data-theme', dark ? 'dark' : 'light') @@ -178,10 +182,11 @@ export function AppShell() { setView('calibration')} onOpenMilestone={() => setView('milestone')} + calibration={calibration ? { n: calibration.model.n, coldStart: calibration.model.coldStart } : undefined} /> ) case 'calibration': - return setView('runway')} /> + return setView('runway')} data={calibration?.data} /> case 'milestone': return setView('runway')} onOpenIssue={openIssue} /> case 'inbox': diff --git a/apps/desktop/src/renderer/src/lib/backlog.ts b/apps/desktop/src/renderer/src/lib/backlog.ts index 4c35e0c..5aec932 100644 --- a/apps/desktop/src/renderer/src/lib/backlog.ts +++ b/apps/desktop/src/renderer/src/lib/backlog.ts @@ -1,17 +1,24 @@ import { + type CalibrationModel, + type CalibrationSample, + calibrationSamples, + COLD_START_THRESHOLD, type DependencyEdge, + fitCalibration, forecast, type GiteaIssue, inferLifecycle, type LifecycleColumn, type LifecycleEvent, type LifecycleInference, + PRIOR_BUCKETS, schedule, type ScheduledItem, selectFocus, + toDurationModel, } from '@commitea/core' -import { type BoardColumn, type BoardIssue, type FocusIssue } from '../data/fixtures.js' +import { type BoardColumn, type BoardIssue, type CalibrationData, type FocusIssue } from '../data/fixtures.js' import { type BurnUpData, buildBurnUpData } from './dates.js' type Timelines = Record @@ -100,22 +107,109 @@ export interface ForecastView { cone: BurnUpData p80Label: string rangeLabel: string + /** true while the forecast still runs on code priors (calibration not yet trusted). */ + coldStart: boolean + /** Closed-with-estimate issues feeding calibration so far. */ + calibratedN: number } /** * Monte Carlo forecast over the open backlog, mapped onto a calendar-anchored - * burn-up cone. Returns null when there's nothing to forecast (no open scope) — - * the UI then falls back to the demo cone. `today` is injectable for tests. + * burn-up cone. When a calibration model is supplied and past cold-start, its + * fitted params drive the sim. Returns null when there's nothing to forecast. + * `today` is injectable for tests. */ export function forecastBacklog( issues: GiteaIssue[], deps: DependencyEdge[], today: Date = new Date(), + calibration?: CalibrationModel, ): ForecastView | null { - const f = forecast(toSchedulable(issues), deps) + const model = calibration ? toDurationModel(calibration) : undefined + const f = forecast(toSchedulable(issues), deps, model ? { model } : {}) const cone = buildBurnUpData(f, today) if (!cone) return null - return { scope: f.scope, cone, p80Label: cone.p80Label, rangeLabel: cone.rangeLabel } + return { + scope: f.scope, + cone, + p80Label: cone.p80Label, + rangeLabel: cone.rangeLabel, + coldStart: f.coldStart, + calibratedN: calibration?.n ?? 0, + } +} + +/** Fit the calibration model from the closed backlog's inferred actuals (#1). */ +export function calibrateBacklog( + issues: GiteaIssue[], + timelines: Timelines = {}, + asOf: Date = new Date(), +): CalibrationModel { + return fitCalibration(calibrationSamples(issues, timelines, asOf)) +} + +/** Calibration model + its screen view in one pass over the closed backlog. */ +export function backlogCalibration( + issues: GiteaIssue[], + timelines: Timelines = {}, + asOf: Date = new Date(), +): { model: CalibrationModel; data: CalibrationData } { + const samples = calibrationSamples(issues, timelines, asOf) + const model = fitCalibration(samples) + return { model, data: calibrationData(model, samples, issues) } +} + +const pctFromMu = (mu: number) => Math.round((Math.exp(mu) - 1) * 100) + +/** + * Shape the calibration model + its samples into the screen's view. Buckets and + * people only earn a bias once their sample clears the fit floor; everything + * degrades honestly on a thin (cold-start) dataset. + */ +export function calibrationData( + model: CalibrationModel, + samples: CalibrationSample[], + openIssues: GiteaIssue[], +): CalibrationData { + const labels = PRIOR_BUCKETS.map((b) => { + const inBucket = samples.filter((s) => s.bucket === b) + const fit = model.byBucket[b] + const mu = fit ? fit.mu : model.global.mu + return { + label: `est/${b}d`, + n: fit ? fit.n : inBucket.length, + median: inBucket.length ? `${(b * Math.exp(mu)).toFixed(1)}d` : '—', + bias: fit ? pctFromMu(fit.mu) : null, + } + }) + + const people = Object.entries(model.byPerson).map(([who, pb]) => ({ + who, + n: pb.n, + bias: pctFromMu(model.global.mu + pb.biasMu), + note: '', + })) + + const openEst = openIssues + .filter((i) => i.state === 'open') + .reduce((sum, i) => sum + (i.facts.estimateDays ?? 2), 0) + const effect = model.coldStart + ? { raw: `${model.n}/${COLD_START_THRESHOLD} estimated closes`, banded: 'cold-start priors', p50: '—' } + : { + raw: `${openEst}d estimated`, + banded: `×${Math.exp(model.global.mu).toFixed(2)} median drift`, + p50: `≈${Math.round(openEst * Math.exp(model.global.mu))}d`, + } + + return { + n: model.n, + active: !model.coldStart, + labels, + people, + scatter: samples.map((s) => [s.estimateDays, s.actualWorkingDays]), + fit: Number(Math.exp(model.global.mu).toFixed(2)), + effect, + } } /** diff --git a/packages/core/src/calibration/calibration-v0.test.ts b/packages/core/src/calibration/calibration-v0.test.ts new file mode 100644 index 0000000..09e8486 --- /dev/null +++ b/packages/core/src/calibration/calibration-v0.test.ts @@ -0,0 +1,121 @@ +import { describe, expect, it } from 'vitest' + +import { extractLabelFacts } from '../labels/label-schema.js' +import type { LifecycleEvent } from '../lifecycle/lifecycle-v0.js' +import type { GiteaIssue } from '../gitea/types.js' +import { + CALIBRATION_BUCKET_FLOOR, + calibrationSamples, + type CalibrationSample, + COLD_START_THRESHOLD, + fitCalibration, + toDurationModel, +} from './calibration-v0.js' + +function sample(over: Partial = {}): CalibrationSample { + return { issue: 1, estimateDays: 2, actualWorkingDays: 2, bucket: 2, person: null, ...over } +} + +describe('fitCalibration', () => { + it('is cold-start below the threshold and reports the honest n', () => { + const m = fitCalibration([sample(), sample({ actualWorkingDays: 4 })]) + expect(m.n).toBe(2) + expect(m.coldStart).toBe(true) + }) + + it('flips off cold-start at the threshold', () => { + const many = Array.from({ length: COLD_START_THRESHOLD }, (_, i) => + sample({ issue: i, estimateDays: 2, actualWorkingDays: 3, bucket: 2 }), + ) + const m = fitCalibration(many) + expect(m.n).toBe(COLD_START_THRESHOLD) + expect(m.coldStart).toBe(false) + }) + + it('recovers the global median ratio (mu = mean log-ratio)', () => { + // every actual is exactly 2x its estimate → mu = ln 2 + const m = fitCalibration(Array.from({ length: 25 }, (_, i) => sample({ issue: i, estimateDays: 2, actualWorkingDays: 4 }))) + expect(m.global.mu).toBeCloseTo(Math.log(2), 6) + }) + + it('fits a bucket only once it clears the floor', () => { + const twos = Array.from({ length: CALIBRATION_BUCKET_FLOOR }, (_, i) => + sample({ issue: i, estimateDays: 2, actualWorkingDays: 3, bucket: 2 }), + ) + const oneThin = [sample({ issue: 99, estimateDays: 5, actualWorkingDays: 9, bucket: 5 })] + const m = fitCalibration([...twos, ...oneThin]) + expect(m.byBucket[2]?.n).toBe(CALIBRATION_BUCKET_FLOOR) + expect(m.byBucket[5]).toBeUndefined() // only 1 sample, below floor + }) + + it('drops non-positive estimates/actuals', () => { + const m = fitCalibration([sample({ actualWorkingDays: 0 }), sample({ estimateDays: 0 }), sample()]) + expect(m.n).toBe(1) + }) + + it('derives a per-person bias relative to global', () => { + // one person consistently runs longer than the mean + const base = Array.from({ length: 20 }, (_, i) => sample({ issue: i, actualWorkingDays: 2, person: 'ak' })) + const slow = Array.from({ length: 3 }, (_, i) => sample({ issue: 100 + i, actualWorkingDays: 6, person: 'sm' })) + const m = fitCalibration([...base, ...slow]) + expect(m.byPerson['sm'].biasMu).toBeGreaterThan(0) + expect(m.byPerson['ak'].biasMu).toBeLessThan(0) + }) +}) + +describe('toDurationModel', () => { + it('projects the fit down to the params forecast needs', () => { + const m = fitCalibration( + Array.from({ length: 25 }, (_, i) => sample({ issue: i, estimateDays: 2, actualWorkingDays: 3, bucket: 2 })), + ) + const dm = toDurationModel(m) + expect(dm.coldStart).toBe(false) + expect(dm.byBucket[2].mu).toBeCloseTo(m.byBucket[2].mu, 6) + expect(dm.global.mu).toBeCloseTo(m.global.mu, 6) + }) +}) + +describe('calibrationSamples', () => { + const asOf = new Date('2026-02-01T00:00:00Z') + + function issue(over: Partial): GiteaIssue { + const labels = over.labels ?? [] + return { + number: 1, + title: '#1', + body: '', + state: 'closed', + labels, + facts: extractLabelFacts(labels), + milestone: null, + assignee: null, + assignees: [], + createdAt: '2026-01-05T09:00:00Z', + updatedAt: '2026-01-12T09:00:00Z', + closedAt: '2026-01-12T09:00:00Z', + url: '', + ...over, + } + } + + const events = (i: number): Record => ({ + [i]: [{ type: 'commit', at: '2026-01-07T09:00:00Z' }, { type: 'close', at: '2026-01-12T09:00:00Z' }], + }) + + it('samples closed, estimated issues with a resolvable actual', () => { + const i = issue({ number: 7, labels: ['est/2d'], assignee: 'sm', closedAt: '2026-01-12T09:00:00Z' }) + const [s] = calibrationSamples([i], events(7), asOf) + expect(s.issue).toBe(7) + expect(s.estimateDays).toBe(2) + expect(s.bucket).toBe(2) + expect(s.person).toBe('sm') + // Wed 2026-01-07 → Mon 2026-01-12 = Wed,Thu,Fri = 3 working days + expect(s.actualWorkingDays).toBe(3) + }) + + it('skips open issues and closed ones without an estimate', () => { + const open = issue({ number: 8, state: 'open', labels: ['est/2d'], closedAt: null }) + const noEst = issue({ number: 9, labels: [] }) + expect(calibrationSamples([open, noEst], { ...events(8), ...events(9) }, asOf)).toEqual([]) + }) +}) diff --git a/packages/core/src/calibration/calibration-v0.ts b/packages/core/src/calibration/calibration-v0.ts new file mode 100644 index 0000000..75ec6fb --- /dev/null +++ b/packages/core/src/calibration/calibration-v0.ts @@ -0,0 +1,127 @@ +/** + * Calibration, v0 — fit the team's own estimate-vs-actual history so the + * forecast stops guessing (D3). The "actual" is the working time lifecycle + * inference derives from git events (#5), never manual tracking. Fit a + * lognormal on log(actual / estimate) globally and per estimate bucket; until + * the sample clears the cold-start threshold, the forecast keeps using the + * code-resident priors and this model just reports progress toward it. + */ + +import { type DurationModel, type LognormalPrior, nearestBucket } from '../forecast/forecast-v0.js' +import { inferLifecycle, type LifecycleEvent } from '../lifecycle/lifecycle-v0.js' +import type { GiteaIssue } from '../gitea/types.js' + +/** Global sample size at which the fit takes over from the cold-start priors. */ +export const COLD_START_THRESHOLD = 20 +/** Minimum per-bucket sample before that bucket earns its own fit. */ +export const CALIBRATION_BUCKET_FLOOR = 3 +/** Fallback spread when a group is too small to estimate one. */ +const DEFAULT_SIGMA = 0.4 + +/** One closed issue's estimate vs its inferred actual. */ +export interface CalibrationSample { + issue: number + estimateDays: number + actualWorkingDays: number + bucket: number + person: string | null +} + +export interface BucketFit extends LognormalPrior { + n: number +} + +export interface PersonBias { + /** Additive to global mu (log space). */ + biasMu: number + n: number +} + +export interface CalibrationModel { + /** Closed issues with an estimate + a resolvable actual. */ + n: number + /** true while n < COLD_START_THRESHOLD — forecast keeps the code priors. */ + coldStart: boolean + global: LognormalPrior + byBucket: Record + byPerson: Record +} + +function mean(xs: number[]): number { + return xs.reduce((a, b) => a + b, 0) / xs.length +} + +/** Sample standard deviation; falls back to DEFAULT_SIGMA below 2 points. */ +function stddev(xs: number[], mu: number): number { + if (xs.length < 2) return DEFAULT_SIGMA + const variance = xs.reduce((a, x) => a + (x - mu) ** 2, 0) / (xs.length - 1) + return Math.sqrt(variance) || DEFAULT_SIGMA +} + +/** Fit a calibration model from estimate-vs-actual samples. Pure. */ +export function fitCalibration(samples: CalibrationSample[]): CalibrationModel { + const usable = samples.filter((s) => s.estimateDays > 0 && s.actualWorkingDays > 0) + const n = usable.length + const coldStart = n < COLD_START_THRESHOLD + + const logRatios = usable.map((s) => Math.log(s.actualWorkingDays / s.estimateDays)) + const globalMu = n ? mean(logRatios) : 0 + const global: LognormalPrior = { mu: globalMu, sigma: n ? stddev(logRatios, globalMu) : DEFAULT_SIGMA } + + const byBucket: Record = {} + const byPerson: Record = {} + const groups = new Map() + const people = new Map() + for (const s of usable) { + const lr = Math.log(s.actualWorkingDays / s.estimateDays) + ;(groups.get(s.bucket) ?? groups.set(s.bucket, []).get(s.bucket)!).push(lr) + if (s.person) (people.get(s.person) ?? people.set(s.person, []).get(s.person)!).push(lr) + } + for (const [bucket, lrs] of groups) { + if (lrs.length < CALIBRATION_BUCKET_FLOOR) continue + const mu = mean(lrs) + byBucket[bucket] = { mu, sigma: stddev(lrs, mu), n: lrs.length } + } + for (const [person, lrs] of people) { + if (lrs.length < CALIBRATION_BUCKET_FLOOR) continue + byPerson[person] = { biasMu: mean(lrs) - globalMu, n: lrs.length } + } + + return { n, coldStart, global, byBucket, byPerson } +} + +/** The subset of a model `forecast` consumes. */ +export function toDurationModel(model: CalibrationModel): DurationModel { + const byBucket: Record = {} + for (const [bucket, fit] of Object.entries(model.byBucket)) { + byBucket[Number(bucket)] = { mu: fit.mu, sigma: fit.sigma } + } + return { coldStart: model.coldStart, global: model.global, byBucket } +} + +/** + * Extract calibration samples from the closed backlog: each closed issue that + * carries an estimate and yields an inferred actual working duration. + */ +export function calibrationSamples( + issues: GiteaIssue[], + timelines: Record, + asOf: Date, +): CalibrationSample[] { + const out: CalibrationSample[] = [] + for (const issue of issues) { + if (issue.state !== 'closed') continue + const estimateDays = issue.facts.estimateDays + if (estimateDays == null) continue + const inf = inferLifecycle(issue, timelines[issue.number] ?? [], asOf) + if (inf.actualWorkingDays == null || inf.actualWorkingDays <= 0) continue + out.push({ + issue: issue.number, + estimateDays, + actualWorkingDays: inf.actualWorkingDays, + bucket: nearestBucket(estimateDays), + person: issue.assignee, + }) + } + return out +} diff --git a/packages/core/src/forecast/forecast-v0.test.ts b/packages/core/src/forecast/forecast-v0.test.ts index 7919e68..17c9d14 100644 --- a/packages/core/src/forecast/forecast-v0.test.ts +++ b/packages/core/src/forecast/forecast-v0.test.ts @@ -97,4 +97,27 @@ describe('forecast', () => { expect(f.scope).toBe(2) expect(f.p50Day).toBeGreaterThan(0) }) + + it('a cold-start model changes nothing — the code priors still drive it', () => { + const priors = forecast(scope, [], { trials: 1000, seed: 7 }) + const cold = forecast(scope, [], { + trials: 1000, + seed: 7, + model: { coldStart: true, global: { mu: 5, sigma: 0.1 }, byBucket: {} }, + }) + expect(cold.coldStart).toBe(true) + expect(cold.p50Day).toBeCloseTo(priors.p50Day, 6) + }) + + it('a fitted model drives the sim once past cold-start', () => { + // an optimistic fit (mu < 0, tight sigma) should land the scope sooner than the pessimistic priors + const priors = forecast(scope, [], { trials: 2000, seed: 7 }) + const fitted = forecast(scope, [], { + trials: 2000, + seed: 7, + model: { coldStart: false, global: { mu: -0.2, sigma: 0.1 }, byBucket: {} }, + }) + expect(fitted.coldStart).toBe(false) + expect(fitted.p50Day).toBeLessThan(priors.p50Day) + }) }) diff --git a/packages/core/src/forecast/forecast-v0.ts b/packages/core/src/forecast/forecast-v0.ts index 82308c0..04134fb 100644 --- a/packages/core/src/forecast/forecast-v0.ts +++ b/packages/core/src/forecast/forecast-v0.ts @@ -41,15 +41,29 @@ export const COLD_START_PRIORS: Record = { 8: { mu: 0.16, sigma: 0.36 }, } -const PRIOR_BUCKETS = [1, 2, 3, 5, 8] +export const PRIOR_BUCKETS = [1, 2, 3, 5, 8] /** Nearest estimate bucket (ties resolve to the smaller bucket). */ -export function priorForEstimate(days: number): LognormalPrior { +export function nearestBucket(days: number): number { let best = PRIOR_BUCKETS[0] for (const b of PRIOR_BUCKETS) { if (Math.abs(b - days) < Math.abs(best - days)) best = b } - return COLD_START_PRIORS[best] + return best +} + +/** The cold-start prior for the bucket nearest to `days`. */ +export function priorForEstimate(days: number): LognormalPrior { + return COLD_START_PRIORS[nearestBucket(days)] +} + +/** The lognormal parameters `forecast` needs, per estimate bucket. */ +export interface DurationModel { + coldStart: boolean + /** Fallback params (used when a bucket lacks its own fit). */ + global: LognormalPrior + /** Per-bucket fitted params; missing buckets fall back to `global`. */ + byBucket: Record } export interface ForecastOptions { @@ -57,6 +71,19 @@ export interface ForecastOptions { trials?: number /** PRNG seed. Fixed by default so a forecast is reproducible. */ seed?: number + /** + * Fitted duration model. When present and not cold-start, its params drive + * the sim; otherwise the code-resident cold-start priors do. + */ + model?: DurationModel +} + +/** Resolve the lognormal params for an estimate, preferring a fitted model. */ +export function durationParams(days: number, model?: DurationModel): LognormalPrior { + if (model && !model.coldStart) { + return model.byBucket[nearestBucket(days)] ?? model.global + } + return priorForEstimate(days) } export interface BurnUpPoint { @@ -121,13 +148,14 @@ export function forecast( ): Forecast { const trials = options.trials ?? DEFAULT_TRIALS const seed = options.seed ?? DEFAULT_SEED + const coldStart = options.model ? options.model.coldStart : true const order = schedule(issues, edges).items // empty when a dependency cycle exists const n = order.length if (n === 0) { - return { scope: 0, trials, coldStart: true, p50Day: 0, p80Day: 0, p95Day: 0, curve: [] } + return { scope: 0, trials, coldStart, p50Day: 0, p80Day: 0, p95Day: 0, curve: [] } } - const priors = order.map((it) => priorForEstimate(it.durationDays)) + const priors = order.map((it) => durationParams(it.durationDays, options.model)) const rng = mulberry32(seed) // endByRank[k][t] = working day the (k+1)-th scheduled issue completes on trial t. @@ -156,7 +184,7 @@ export function forecast( return { scope: n, trials, - coldStart: true, + coldStart, p50Day: percentile(total, 0.5), p80Day: percentile(total, 0.8), p95Day: percentile(total, 0.95), diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index 2a097f0..c5b9716 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -45,5 +45,32 @@ export type { SchedulePlan, } from './scheduler/scheduler-v0.js' -export { COLD_START_PRIORS, forecast, priorForEstimate } from './forecast/forecast-v0.js' -export type { BurnUpPoint, Forecast, ForecastOptions, LognormalPrior } from './forecast/forecast-v0.js' +export { + COLD_START_PRIORS, + durationParams, + forecast, + nearestBucket, + PRIOR_BUCKETS, + priorForEstimate, +} from './forecast/forecast-v0.js' +export type { + BurnUpPoint, + DurationModel, + Forecast, + ForecastOptions, + LognormalPrior, +} from './forecast/forecast-v0.js' + +export { + CALIBRATION_BUCKET_FLOOR, + calibrationSamples, + COLD_START_THRESHOLD, + fitCalibration, + toDurationModel, +} from './calibration/calibration-v0.js' +export type { + BucketFit, + CalibrationModel, + CalibrationSample, + PersonBias, +} from './calibration/calibration-v0.js'