feat: calibration from closed-issue actuals → forecast flips off cold-start (#1)
Close the D3 loop. The forecast now learns from the team's own estimate-vs-actual history (the working time #5 infers from git events) instead of guessing forever. core (@commitea/core/calibration-v0): - fitCalibration(samples): lognormal fit on log(actual/estimate) — global + per-bucket (once a bucket clears the floor) + per-person bias. coldStart until n >= 20 closed-with-estimate issues. - calibrationSamples(): pull those samples from the closed backlog via lifecycle inference (estimate label vs inferred actualWorkingDays). - toDurationModel(): project the fit to the params forecast consumes. - forecast() gains options.model: when past cold-start, fitted params drive the sim (per bucket, global fallback); otherwise the code priors do. Forecast.coldStart now reflects the model. nearestBucket extracted + exported. app: - AppShell fits calibration once from the reconciled backlog, feeds the model into forecastBacklog (cone), and drives the Calibration screen + Runway header. - Focus cone footer, Runway note, and Calibration screen now say cold-start (N/20) vs calibrated (on N closed) from real data; Calibration scatter / bucket bias / per-person all fitted, degrading honestly on a thin dataset. Known refinement: same-day closes yield 0 working-day actuals (day-granular) and are excluded, so a fast-moving repo can sit at n=0 — honest, but a fractional (hours-based) actual would let those count. Per-person uses gitea login, not display name, until the person map lands. Note: also re-lands #10 (Monte Carlo) and #5 (lifecycle) which merged into their stacked base branches but never propagated to main (stacked-merge trap); this branch is cut from main and carries all three so main is whole again. Verified: 74 core tests green (9 calibration + 2 forecast-switch added), desktop typecheck clean, 14 fixture e2e green, live spec asserts the real cold-start calibration surface (Runway note + screen badge fitted from actuals). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
127
packages/core/src/calibration/calibration-v0.ts
Normal file
127
packages/core/src/calibration/calibration-v0.ts
Normal file
@@ -0,0 +1,127 @@
|
||||
/**
|
||||
* Calibration, v0 — fit the team's own estimate-vs-actual history so the
|
||||
* forecast stops guessing (D3). The "actual" is the working time lifecycle
|
||||
* inference derives from git events (#5), never manual tracking. Fit a
|
||||
* lognormal on log(actual / estimate) globally and per estimate bucket; until
|
||||
* the sample clears the cold-start threshold, the forecast keeps using the
|
||||
* code-resident priors and this model just reports progress toward it.
|
||||
*/
|
||||
|
||||
import { type DurationModel, type LognormalPrior, nearestBucket } from '../forecast/forecast-v0.js'
|
||||
import { inferLifecycle, type LifecycleEvent } from '../lifecycle/lifecycle-v0.js'
|
||||
import type { GiteaIssue } from '../gitea/types.js'
|
||||
|
||||
/** Global sample size at which the fit takes over from the cold-start priors. */
|
||||
export const COLD_START_THRESHOLD = 20
|
||||
/** Minimum per-bucket sample before that bucket earns its own fit. */
|
||||
export const CALIBRATION_BUCKET_FLOOR = 3
|
||||
/** Fallback spread when a group is too small to estimate one. */
|
||||
const DEFAULT_SIGMA = 0.4
|
||||
|
||||
/** One closed issue's estimate vs its inferred actual. */
|
||||
export interface CalibrationSample {
|
||||
issue: number
|
||||
estimateDays: number
|
||||
actualWorkingDays: number
|
||||
bucket: number
|
||||
person: string | null
|
||||
}
|
||||
|
||||
export interface BucketFit extends LognormalPrior {
|
||||
n: number
|
||||
}
|
||||
|
||||
export interface PersonBias {
|
||||
/** Additive to global mu (log space). */
|
||||
biasMu: number
|
||||
n: number
|
||||
}
|
||||
|
||||
export interface CalibrationModel {
|
||||
/** Closed issues with an estimate + a resolvable actual. */
|
||||
n: number
|
||||
/** true while n < COLD_START_THRESHOLD — forecast keeps the code priors. */
|
||||
coldStart: boolean
|
||||
global: LognormalPrior
|
||||
byBucket: Record<number, BucketFit>
|
||||
byPerson: Record<string, PersonBias>
|
||||
}
|
||||
|
||||
function mean(xs: number[]): number {
|
||||
return xs.reduce((a, b) => a + b, 0) / xs.length
|
||||
}
|
||||
|
||||
/** Sample standard deviation; falls back to DEFAULT_SIGMA below 2 points. */
|
||||
function stddev(xs: number[], mu: number): number {
|
||||
if (xs.length < 2) return DEFAULT_SIGMA
|
||||
const variance = xs.reduce((a, x) => a + (x - mu) ** 2, 0) / (xs.length - 1)
|
||||
return Math.sqrt(variance) || DEFAULT_SIGMA
|
||||
}
|
||||
|
||||
/** Fit a calibration model from estimate-vs-actual samples. Pure. */
|
||||
export function fitCalibration(samples: CalibrationSample[]): CalibrationModel {
|
||||
const usable = samples.filter((s) => s.estimateDays > 0 && s.actualWorkingDays > 0)
|
||||
const n = usable.length
|
||||
const coldStart = n < COLD_START_THRESHOLD
|
||||
|
||||
const logRatios = usable.map((s) => Math.log(s.actualWorkingDays / s.estimateDays))
|
||||
const globalMu = n ? mean(logRatios) : 0
|
||||
const global: LognormalPrior = { mu: globalMu, sigma: n ? stddev(logRatios, globalMu) : DEFAULT_SIGMA }
|
||||
|
||||
const byBucket: Record<number, BucketFit> = {}
|
||||
const byPerson: Record<string, PersonBias> = {}
|
||||
const groups = new Map<number, number[]>()
|
||||
const people = new Map<string, number[]>()
|
||||
for (const s of usable) {
|
||||
const lr = Math.log(s.actualWorkingDays / s.estimateDays)
|
||||
;(groups.get(s.bucket) ?? groups.set(s.bucket, []).get(s.bucket)!).push(lr)
|
||||
if (s.person) (people.get(s.person) ?? people.set(s.person, []).get(s.person)!).push(lr)
|
||||
}
|
||||
for (const [bucket, lrs] of groups) {
|
||||
if (lrs.length < CALIBRATION_BUCKET_FLOOR) continue
|
||||
const mu = mean(lrs)
|
||||
byBucket[bucket] = { mu, sigma: stddev(lrs, mu), n: lrs.length }
|
||||
}
|
||||
for (const [person, lrs] of people) {
|
||||
if (lrs.length < CALIBRATION_BUCKET_FLOOR) continue
|
||||
byPerson[person] = { biasMu: mean(lrs) - globalMu, n: lrs.length }
|
||||
}
|
||||
|
||||
return { n, coldStart, global, byBucket, byPerson }
|
||||
}
|
||||
|
||||
/** The subset of a model `forecast` consumes. */
|
||||
export function toDurationModel(model: CalibrationModel): DurationModel {
|
||||
const byBucket: Record<number, LognormalPrior> = {}
|
||||
for (const [bucket, fit] of Object.entries(model.byBucket)) {
|
||||
byBucket[Number(bucket)] = { mu: fit.mu, sigma: fit.sigma }
|
||||
}
|
||||
return { coldStart: model.coldStart, global: model.global, byBucket }
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract calibration samples from the closed backlog: each closed issue that
|
||||
* carries an estimate and yields an inferred actual working duration.
|
||||
*/
|
||||
export function calibrationSamples(
|
||||
issues: GiteaIssue[],
|
||||
timelines: Record<number, LifecycleEvent[]>,
|
||||
asOf: Date,
|
||||
): CalibrationSample[] {
|
||||
const out: CalibrationSample[] = []
|
||||
for (const issue of issues) {
|
||||
if (issue.state !== 'closed') continue
|
||||
const estimateDays = issue.facts.estimateDays
|
||||
if (estimateDays == null) continue
|
||||
const inf = inferLifecycle(issue, timelines[issue.number] ?? [], asOf)
|
||||
if (inf.actualWorkingDays == null || inf.actualWorkingDays <= 0) continue
|
||||
out.push({
|
||||
issue: issue.number,
|
||||
estimateDays,
|
||||
actualWorkingDays: inf.actualWorkingDays,
|
||||
bucket: nearestBucket(estimateDays),
|
||||
person: issue.assignee,
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
Reference in New Issue
Block a user