feat: calibration from closed-issue actuals → forecast flips off cold-start (#1)
Close the D3 loop. The forecast now learns from the team's own estimate-vs-actual history (the working time #5 infers from git events) instead of guessing forever. core (@commitea/core/calibration-v0): - fitCalibration(samples): lognormal fit on log(actual/estimate) — global + per-bucket (once a bucket clears the floor) + per-person bias. coldStart until n >= 20 closed-with-estimate issues. - calibrationSamples(): pull those samples from the closed backlog via lifecycle inference (estimate label vs inferred actualWorkingDays). - toDurationModel(): project the fit to the params forecast consumes. - forecast() gains options.model: when past cold-start, fitted params drive the sim (per bucket, global fallback); otherwise the code priors do. Forecast.coldStart now reflects the model. nearestBucket extracted + exported. app: - AppShell fits calibration once from the reconciled backlog, feeds the model into forecastBacklog (cone), and drives the Calibration screen + Runway header. - Focus cone footer, Runway note, and Calibration screen now say cold-start (N/20) vs calibrated (on N closed) from real data; Calibration scatter / bucket bias / per-person all fitted, degrading honestly on a thin dataset. Known refinement: same-day closes yield 0 working-day actuals (day-granular) and are excluded, so a fast-moving repo can sit at n=0 — honest, but a fractional (hours-based) actual would let those count. Per-person uses gitea login, not display name, until the person map lands. Note: also re-lands #10 (Monte Carlo) and #5 (lifecycle) which merged into their stacked base branches but never propagated to main (stacked-merge trap); this branch is cut from main and carries all three so main is whole again. Verified: 74 core tests green (9 calibration + 2 forecast-switch added), desktop typecheck clean, 14 fixture e2e green, live spec asserts the real cold-start calibration surface (Runway note + screen badge fitted from actuals). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
121
packages/core/src/calibration/calibration-v0.test.ts
Normal file
121
packages/core/src/calibration/calibration-v0.test.ts
Normal file
@@ -0,0 +1,121 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
|
||||
import { extractLabelFacts } from '../labels/label-schema.js'
|
||||
import type { LifecycleEvent } from '../lifecycle/lifecycle-v0.js'
|
||||
import type { GiteaIssue } from '../gitea/types.js'
|
||||
import {
|
||||
CALIBRATION_BUCKET_FLOOR,
|
||||
calibrationSamples,
|
||||
type CalibrationSample,
|
||||
COLD_START_THRESHOLD,
|
||||
fitCalibration,
|
||||
toDurationModel,
|
||||
} from './calibration-v0.js'
|
||||
|
||||
function sample(over: Partial<CalibrationSample> = {}): CalibrationSample {
|
||||
return { issue: 1, estimateDays: 2, actualWorkingDays: 2, bucket: 2, person: null, ...over }
|
||||
}
|
||||
|
||||
describe('fitCalibration', () => {
|
||||
it('is cold-start below the threshold and reports the honest n', () => {
|
||||
const m = fitCalibration([sample(), sample({ actualWorkingDays: 4 })])
|
||||
expect(m.n).toBe(2)
|
||||
expect(m.coldStart).toBe(true)
|
||||
})
|
||||
|
||||
it('flips off cold-start at the threshold', () => {
|
||||
const many = Array.from({ length: COLD_START_THRESHOLD }, (_, i) =>
|
||||
sample({ issue: i, estimateDays: 2, actualWorkingDays: 3, bucket: 2 }),
|
||||
)
|
||||
const m = fitCalibration(many)
|
||||
expect(m.n).toBe(COLD_START_THRESHOLD)
|
||||
expect(m.coldStart).toBe(false)
|
||||
})
|
||||
|
||||
it('recovers the global median ratio (mu = mean log-ratio)', () => {
|
||||
// every actual is exactly 2x its estimate → mu = ln 2
|
||||
const m = fitCalibration(Array.from({ length: 25 }, (_, i) => sample({ issue: i, estimateDays: 2, actualWorkingDays: 4 })))
|
||||
expect(m.global.mu).toBeCloseTo(Math.log(2), 6)
|
||||
})
|
||||
|
||||
it('fits a bucket only once it clears the floor', () => {
|
||||
const twos = Array.from({ length: CALIBRATION_BUCKET_FLOOR }, (_, i) =>
|
||||
sample({ issue: i, estimateDays: 2, actualWorkingDays: 3, bucket: 2 }),
|
||||
)
|
||||
const oneThin = [sample({ issue: 99, estimateDays: 5, actualWorkingDays: 9, bucket: 5 })]
|
||||
const m = fitCalibration([...twos, ...oneThin])
|
||||
expect(m.byBucket[2]?.n).toBe(CALIBRATION_BUCKET_FLOOR)
|
||||
expect(m.byBucket[5]).toBeUndefined() // only 1 sample, below floor
|
||||
})
|
||||
|
||||
it('drops non-positive estimates/actuals', () => {
|
||||
const m = fitCalibration([sample({ actualWorkingDays: 0 }), sample({ estimateDays: 0 }), sample()])
|
||||
expect(m.n).toBe(1)
|
||||
})
|
||||
|
||||
it('derives a per-person bias relative to global', () => {
|
||||
// one person consistently runs longer than the mean
|
||||
const base = Array.from({ length: 20 }, (_, i) => sample({ issue: i, actualWorkingDays: 2, person: 'ak' }))
|
||||
const slow = Array.from({ length: 3 }, (_, i) => sample({ issue: 100 + i, actualWorkingDays: 6, person: 'sm' }))
|
||||
const m = fitCalibration([...base, ...slow])
|
||||
expect(m.byPerson['sm'].biasMu).toBeGreaterThan(0)
|
||||
expect(m.byPerson['ak'].biasMu).toBeLessThan(0)
|
||||
})
|
||||
})
|
||||
|
||||
describe('toDurationModel', () => {
|
||||
it('projects the fit down to the params forecast needs', () => {
|
||||
const m = fitCalibration(
|
||||
Array.from({ length: 25 }, (_, i) => sample({ issue: i, estimateDays: 2, actualWorkingDays: 3, bucket: 2 })),
|
||||
)
|
||||
const dm = toDurationModel(m)
|
||||
expect(dm.coldStart).toBe(false)
|
||||
expect(dm.byBucket[2].mu).toBeCloseTo(m.byBucket[2].mu, 6)
|
||||
expect(dm.global.mu).toBeCloseTo(m.global.mu, 6)
|
||||
})
|
||||
})
|
||||
|
||||
describe('calibrationSamples', () => {
|
||||
const asOf = new Date('2026-02-01T00:00:00Z')
|
||||
|
||||
function issue(over: Partial<GiteaIssue>): GiteaIssue {
|
||||
const labels = over.labels ?? []
|
||||
return {
|
||||
number: 1,
|
||||
title: '#1',
|
||||
body: '',
|
||||
state: 'closed',
|
||||
labels,
|
||||
facts: extractLabelFacts(labels),
|
||||
milestone: null,
|
||||
assignee: null,
|
||||
assignees: [],
|
||||
createdAt: '2026-01-05T09:00:00Z',
|
||||
updatedAt: '2026-01-12T09:00:00Z',
|
||||
closedAt: '2026-01-12T09:00:00Z',
|
||||
url: '',
|
||||
...over,
|
||||
}
|
||||
}
|
||||
|
||||
const events = (i: number): Record<number, LifecycleEvent[]> => ({
|
||||
[i]: [{ type: 'commit', at: '2026-01-07T09:00:00Z' }, { type: 'close', at: '2026-01-12T09:00:00Z' }],
|
||||
})
|
||||
|
||||
it('samples closed, estimated issues with a resolvable actual', () => {
|
||||
const i = issue({ number: 7, labels: ['est/2d'], assignee: 'sm', closedAt: '2026-01-12T09:00:00Z' })
|
||||
const [s] = calibrationSamples([i], events(7), asOf)
|
||||
expect(s.issue).toBe(7)
|
||||
expect(s.estimateDays).toBe(2)
|
||||
expect(s.bucket).toBe(2)
|
||||
expect(s.person).toBe('sm')
|
||||
// Wed 2026-01-07 → Mon 2026-01-12 = Wed,Thu,Fri = 3 working days
|
||||
expect(s.actualWorkingDays).toBe(3)
|
||||
})
|
||||
|
||||
it('skips open issues and closed ones without an estimate', () => {
|
||||
const open = issue({ number: 8, state: 'open', labels: ['est/2d'], closedAt: null })
|
||||
const noEst = issue({ number: 9, labels: [] })
|
||||
expect(calibrationSamples([open, noEst], { ...events(8), ...events(9) }, asOf)).toEqual([])
|
||||
})
|
||||
})
|
||||
Reference in New Issue
Block a user