feat: calibration from closed-issue actuals → forecast flips off cold-start (#1)

Close the D3 loop. The forecast now learns from the team's own estimate-vs-actual
history (the working time #5 infers from git events) instead of guessing forever.

core (@commitea/core/calibration-v0):
- fitCalibration(samples): lognormal fit on log(actual/estimate) — global +
  per-bucket (once a bucket clears the floor) + per-person bias. coldStart until
  n >= 20 closed-with-estimate issues.
- calibrationSamples(): pull those samples from the closed backlog via lifecycle
  inference (estimate label vs inferred actualWorkingDays).
- toDurationModel(): project the fit to the params forecast consumes.
- forecast() gains options.model: when past cold-start, fitted params drive the
  sim (per bucket, global fallback); otherwise the code priors do. Forecast.coldStart
  now reflects the model. nearestBucket extracted + exported.

app:
- AppShell fits calibration once from the reconciled backlog, feeds the model into
  forecastBacklog (cone), and drives the Calibration screen + Runway header.
- Focus cone footer, Runway note, and Calibration screen now say cold-start (N/20)
  vs calibrated (on N closed) from real data; Calibration scatter / bucket bias /
  per-person all fitted, degrading honestly on a thin dataset.

Known refinement: same-day closes yield 0 working-day actuals (day-granular) and
are excluded, so a fast-moving repo can sit at n=0 — honest, but a fractional
(hours-based) actual would let those count. Per-person uses gitea login, not
display name, until the person map lands.

Note: also re-lands #10 (Monte Carlo) and #5 (lifecycle) which merged into their
stacked base branches but never propagated to main (stacked-merge trap); this
branch is cut from main and carries all three so main is whole again.

Verified: 74 core tests green (9 calibration + 2 forecast-switch added), desktop
typecheck clean, 14 fixture e2e green, live spec asserts the real cold-start
calibration surface (Runway note + screen badge fitted from actuals).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Croissant Le Doux
2026-07-08 19:51:50 -04:00
parent 9cedd8646e
commit 7e26de1b6c
11 changed files with 476 additions and 24 deletions

View File

@@ -0,0 +1,121 @@
import { describe, expect, it } from 'vitest'
import { extractLabelFacts } from '../labels/label-schema.js'
import type { LifecycleEvent } from '../lifecycle/lifecycle-v0.js'
import type { GiteaIssue } from '../gitea/types.js'
import {
CALIBRATION_BUCKET_FLOOR,
calibrationSamples,
type CalibrationSample,
COLD_START_THRESHOLD,
fitCalibration,
toDurationModel,
} from './calibration-v0.js'
function sample(over: Partial<CalibrationSample> = {}): CalibrationSample {
return { issue: 1, estimateDays: 2, actualWorkingDays: 2, bucket: 2, person: null, ...over }
}
describe('fitCalibration', () => {
it('is cold-start below the threshold and reports the honest n', () => {
const m = fitCalibration([sample(), sample({ actualWorkingDays: 4 })])
expect(m.n).toBe(2)
expect(m.coldStart).toBe(true)
})
it('flips off cold-start at the threshold', () => {
const many = Array.from({ length: COLD_START_THRESHOLD }, (_, i) =>
sample({ issue: i, estimateDays: 2, actualWorkingDays: 3, bucket: 2 }),
)
const m = fitCalibration(many)
expect(m.n).toBe(COLD_START_THRESHOLD)
expect(m.coldStart).toBe(false)
})
it('recovers the global median ratio (mu = mean log-ratio)', () => {
// every actual is exactly 2x its estimate → mu = ln 2
const m = fitCalibration(Array.from({ length: 25 }, (_, i) => sample({ issue: i, estimateDays: 2, actualWorkingDays: 4 })))
expect(m.global.mu).toBeCloseTo(Math.log(2), 6)
})
it('fits a bucket only once it clears the floor', () => {
const twos = Array.from({ length: CALIBRATION_BUCKET_FLOOR }, (_, i) =>
sample({ issue: i, estimateDays: 2, actualWorkingDays: 3, bucket: 2 }),
)
const oneThin = [sample({ issue: 99, estimateDays: 5, actualWorkingDays: 9, bucket: 5 })]
const m = fitCalibration([...twos, ...oneThin])
expect(m.byBucket[2]?.n).toBe(CALIBRATION_BUCKET_FLOOR)
expect(m.byBucket[5]).toBeUndefined() // only 1 sample, below floor
})
it('drops non-positive estimates/actuals', () => {
const m = fitCalibration([sample({ actualWorkingDays: 0 }), sample({ estimateDays: 0 }), sample()])
expect(m.n).toBe(1)
})
it('derives a per-person bias relative to global', () => {
// one person consistently runs longer than the mean
const base = Array.from({ length: 20 }, (_, i) => sample({ issue: i, actualWorkingDays: 2, person: 'ak' }))
const slow = Array.from({ length: 3 }, (_, i) => sample({ issue: 100 + i, actualWorkingDays: 6, person: 'sm' }))
const m = fitCalibration([...base, ...slow])
expect(m.byPerson['sm'].biasMu).toBeGreaterThan(0)
expect(m.byPerson['ak'].biasMu).toBeLessThan(0)
})
})
describe('toDurationModel', () => {
it('projects the fit down to the params forecast needs', () => {
const m = fitCalibration(
Array.from({ length: 25 }, (_, i) => sample({ issue: i, estimateDays: 2, actualWorkingDays: 3, bucket: 2 })),
)
const dm = toDurationModel(m)
expect(dm.coldStart).toBe(false)
expect(dm.byBucket[2].mu).toBeCloseTo(m.byBucket[2].mu, 6)
expect(dm.global.mu).toBeCloseTo(m.global.mu, 6)
})
})
describe('calibrationSamples', () => {
const asOf = new Date('2026-02-01T00:00:00Z')
function issue(over: Partial<GiteaIssue>): GiteaIssue {
const labels = over.labels ?? []
return {
number: 1,
title: '#1',
body: '',
state: 'closed',
labels,
facts: extractLabelFacts(labels),
milestone: null,
assignee: null,
assignees: [],
createdAt: '2026-01-05T09:00:00Z',
updatedAt: '2026-01-12T09:00:00Z',
closedAt: '2026-01-12T09:00:00Z',
url: '',
...over,
}
}
const events = (i: number): Record<number, LifecycleEvent[]> => ({
[i]: [{ type: 'commit', at: '2026-01-07T09:00:00Z' }, { type: 'close', at: '2026-01-12T09:00:00Z' }],
})
it('samples closed, estimated issues with a resolvable actual', () => {
const i = issue({ number: 7, labels: ['est/2d'], assignee: 'sm', closedAt: '2026-01-12T09:00:00Z' })
const [s] = calibrationSamples([i], events(7), asOf)
expect(s.issue).toBe(7)
expect(s.estimateDays).toBe(2)
expect(s.bucket).toBe(2)
expect(s.person).toBe('sm')
// Wed 2026-01-07 → Mon 2026-01-12 = Wed,Thu,Fri = 3 working days
expect(s.actualWorkingDays).toBe(3)
})
it('skips open issues and closed ones without an estimate', () => {
const open = issue({ number: 8, state: 'open', labels: ['est/2d'], closedAt: null })
const noEst = issue({ number: 9, labels: [] })
expect(calibrationSamples([open, noEst], { ...events(8), ...events(9) }, asOf)).toEqual([])
})
})

View File

@@ -0,0 +1,127 @@
/**
* Calibration, v0 — fit the team's own estimate-vs-actual history so the
* forecast stops guessing (D3). The "actual" is the working time lifecycle
* inference derives from git events (#5), never manual tracking. Fit a
* lognormal on log(actual / estimate) globally and per estimate bucket; until
* the sample clears the cold-start threshold, the forecast keeps using the
* code-resident priors and this model just reports progress toward it.
*/
import { type DurationModel, type LognormalPrior, nearestBucket } from '../forecast/forecast-v0.js'
import { inferLifecycle, type LifecycleEvent } from '../lifecycle/lifecycle-v0.js'
import type { GiteaIssue } from '../gitea/types.js'
/** Global sample size at which the fit takes over from the cold-start priors. */
export const COLD_START_THRESHOLD = 20
/** Minimum per-bucket sample before that bucket earns its own fit. */
export const CALIBRATION_BUCKET_FLOOR = 3
/** Fallback spread when a group is too small to estimate one. */
const DEFAULT_SIGMA = 0.4
/** One closed issue's estimate vs its inferred actual. */
export interface CalibrationSample {
issue: number
estimateDays: number
actualWorkingDays: number
bucket: number
person: string | null
}
export interface BucketFit extends LognormalPrior {
n: number
}
export interface PersonBias {
/** Additive to global mu (log space). */
biasMu: number
n: number
}
export interface CalibrationModel {
/** Closed issues with an estimate + a resolvable actual. */
n: number
/** true while n < COLD_START_THRESHOLD — forecast keeps the code priors. */
coldStart: boolean
global: LognormalPrior
byBucket: Record<number, BucketFit>
byPerson: Record<string, PersonBias>
}
function mean(xs: number[]): number {
return xs.reduce((a, b) => a + b, 0) / xs.length
}
/** Sample standard deviation; falls back to DEFAULT_SIGMA below 2 points. */
function stddev(xs: number[], mu: number): number {
if (xs.length < 2) return DEFAULT_SIGMA
const variance = xs.reduce((a, x) => a + (x - mu) ** 2, 0) / (xs.length - 1)
return Math.sqrt(variance) || DEFAULT_SIGMA
}
/** Fit a calibration model from estimate-vs-actual samples. Pure. */
export function fitCalibration(samples: CalibrationSample[]): CalibrationModel {
const usable = samples.filter((s) => s.estimateDays > 0 && s.actualWorkingDays > 0)
const n = usable.length
const coldStart = n < COLD_START_THRESHOLD
const logRatios = usable.map((s) => Math.log(s.actualWorkingDays / s.estimateDays))
const globalMu = n ? mean(logRatios) : 0
const global: LognormalPrior = { mu: globalMu, sigma: n ? stddev(logRatios, globalMu) : DEFAULT_SIGMA }
const byBucket: Record<number, BucketFit> = {}
const byPerson: Record<string, PersonBias> = {}
const groups = new Map<number, number[]>()
const people = new Map<string, number[]>()
for (const s of usable) {
const lr = Math.log(s.actualWorkingDays / s.estimateDays)
;(groups.get(s.bucket) ?? groups.set(s.bucket, []).get(s.bucket)!).push(lr)
if (s.person) (people.get(s.person) ?? people.set(s.person, []).get(s.person)!).push(lr)
}
for (const [bucket, lrs] of groups) {
if (lrs.length < CALIBRATION_BUCKET_FLOOR) continue
const mu = mean(lrs)
byBucket[bucket] = { mu, sigma: stddev(lrs, mu), n: lrs.length }
}
for (const [person, lrs] of people) {
if (lrs.length < CALIBRATION_BUCKET_FLOOR) continue
byPerson[person] = { biasMu: mean(lrs) - globalMu, n: lrs.length }
}
return { n, coldStart, global, byBucket, byPerson }
}
/** The subset of a model `forecast` consumes. */
export function toDurationModel(model: CalibrationModel): DurationModel {
const byBucket: Record<number, LognormalPrior> = {}
for (const [bucket, fit] of Object.entries(model.byBucket)) {
byBucket[Number(bucket)] = { mu: fit.mu, sigma: fit.sigma }
}
return { coldStart: model.coldStart, global: model.global, byBucket }
}
/**
* Extract calibration samples from the closed backlog: each closed issue that
* carries an estimate and yields an inferred actual working duration.
*/
export function calibrationSamples(
issues: GiteaIssue[],
timelines: Record<number, LifecycleEvent[]>,
asOf: Date,
): CalibrationSample[] {
const out: CalibrationSample[] = []
for (const issue of issues) {
if (issue.state !== 'closed') continue
const estimateDays = issue.facts.estimateDays
if (estimateDays == null) continue
const inf = inferLifecycle(issue, timelines[issue.number] ?? [], asOf)
if (inf.actualWorkingDays == null || inf.actualWorkingDays <= 0) continue
out.push({
issue: issue.number,
estimateDays,
actualWorkingDays: inf.actualWorkingDays,
bucket: nearestBucket(estimateDays),
person: issue.assignee,
})
}
return out
}

View File

@@ -97,4 +97,27 @@ describe('forecast', () => {
expect(f.scope).toBe(2)
expect(f.p50Day).toBeGreaterThan(0)
})
it('a cold-start model changes nothing — the code priors still drive it', () => {
const priors = forecast(scope, [], { trials: 1000, seed: 7 })
const cold = forecast(scope, [], {
trials: 1000,
seed: 7,
model: { coldStart: true, global: { mu: 5, sigma: 0.1 }, byBucket: {} },
})
expect(cold.coldStart).toBe(true)
expect(cold.p50Day).toBeCloseTo(priors.p50Day, 6)
})
it('a fitted model drives the sim once past cold-start', () => {
// an optimistic fit (mu < 0, tight sigma) should land the scope sooner than the pessimistic priors
const priors = forecast(scope, [], { trials: 2000, seed: 7 })
const fitted = forecast(scope, [], {
trials: 2000,
seed: 7,
model: { coldStart: false, global: { mu: -0.2, sigma: 0.1 }, byBucket: {} },
})
expect(fitted.coldStart).toBe(false)
expect(fitted.p50Day).toBeLessThan(priors.p50Day)
})
})

View File

@@ -41,15 +41,29 @@ export const COLD_START_PRIORS: Record<number, LognormalPrior> = {
8: { mu: 0.16, sigma: 0.36 },
}
const PRIOR_BUCKETS = [1, 2, 3, 5, 8]
export const PRIOR_BUCKETS = [1, 2, 3, 5, 8]
/** Nearest estimate bucket (ties resolve to the smaller bucket). */
export function priorForEstimate(days: number): LognormalPrior {
export function nearestBucket(days: number): number {
let best = PRIOR_BUCKETS[0]
for (const b of PRIOR_BUCKETS) {
if (Math.abs(b - days) < Math.abs(best - days)) best = b
}
return COLD_START_PRIORS[best]
return best
}
/** The cold-start prior for the bucket nearest to `days`. */
export function priorForEstimate(days: number): LognormalPrior {
return COLD_START_PRIORS[nearestBucket(days)]
}
/** The lognormal parameters `forecast` needs, per estimate bucket. */
export interface DurationModel {
coldStart: boolean
/** Fallback params (used when a bucket lacks its own fit). */
global: LognormalPrior
/** Per-bucket fitted params; missing buckets fall back to `global`. */
byBucket: Record<number, LognormalPrior>
}
export interface ForecastOptions {
@@ -57,6 +71,19 @@ export interface ForecastOptions {
trials?: number
/** PRNG seed. Fixed by default so a forecast is reproducible. */
seed?: number
/**
* Fitted duration model. When present and not cold-start, its params drive
* the sim; otherwise the code-resident cold-start priors do.
*/
model?: DurationModel
}
/** Resolve the lognormal params for an estimate, preferring a fitted model. */
export function durationParams(days: number, model?: DurationModel): LognormalPrior {
if (model && !model.coldStart) {
return model.byBucket[nearestBucket(days)] ?? model.global
}
return priorForEstimate(days)
}
export interface BurnUpPoint {
@@ -121,13 +148,14 @@ export function forecast(
): Forecast {
const trials = options.trials ?? DEFAULT_TRIALS
const seed = options.seed ?? DEFAULT_SEED
const coldStart = options.model ? options.model.coldStart : true
const order = schedule(issues, edges).items // empty when a dependency cycle exists
const n = order.length
if (n === 0) {
return { scope: 0, trials, coldStart: true, p50Day: 0, p80Day: 0, p95Day: 0, curve: [] }
return { scope: 0, trials, coldStart, p50Day: 0, p80Day: 0, p95Day: 0, curve: [] }
}
const priors = order.map((it) => priorForEstimate(it.durationDays))
const priors = order.map((it) => durationParams(it.durationDays, options.model))
const rng = mulberry32(seed)
// endByRank[k][t] = working day the (k+1)-th scheduled issue completes on trial t.
@@ -156,7 +184,7 @@ export function forecast(
return {
scope: n,
trials,
coldStart: true,
coldStart,
p50Day: percentile(total, 0.5),
p80Day: percentile(total, 0.8),
p95Day: percentile(total, 0.95),

View File

@@ -45,5 +45,32 @@ export type {
SchedulePlan,
} from './scheduler/scheduler-v0.js'
export { COLD_START_PRIORS, forecast, priorForEstimate } from './forecast/forecast-v0.js'
export type { BurnUpPoint, Forecast, ForecastOptions, LognormalPrior } from './forecast/forecast-v0.js'
export {
COLD_START_PRIORS,
durationParams,
forecast,
nearestBucket,
PRIOR_BUCKETS,
priorForEstimate,
} from './forecast/forecast-v0.js'
export type {
BurnUpPoint,
DurationModel,
Forecast,
ForecastOptions,
LognormalPrior,
} from './forecast/forecast-v0.js'
export {
CALIBRATION_BUCKET_FLOOR,
calibrationSamples,
COLD_START_THRESHOLD,
fitCalibration,
toDurationModel,
} from './calibration/calibration-v0.js'
export type {
BucketFit,
CalibrationModel,
CalibrationSample,
PersonBias,
} from './calibration/calibration-v0.js'