Files
commitea/packages/core/src/calibration/calibration-v0.test.ts
Croissant Le Doux 89c873b368 calibration: count same-day closes honestly (#34)
The cold-start surface showed "N/20 closed issues estimated", implying you're
just (20−N) closes away. But calibrationSamples silently drops closed+estimated
issues that closed in 0 working days (same-day closes) — real closes that
structurally can't calibrate. On this repo that's 10 of 24 closes hidden: the
note read 14/20 as if 6 away, when a third of the history will never count.

- core: `calibrationCoverage(issues, timelines, asOf)` → { candidates, usable,
  excludedSameDay }, counting the silently-excluded same-day closes. Pure, tested.
- surface it: CalibrationData gains `excludedSameDay`; backlogCalibration returns
  the coverage; the Runway note and the Calibration screen now say "… · N same-day
  closes can't calibrate" so the thin sample is explained, not just reported.

Verified on christian/commitea: closed=24, usable=14, excludedSameDay=10.
131 core green (incl. new coverage test); core + desktop typecheck; 14 fixture e2e.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-09 15:18:16 -04:00

145 lines
5.8 KiB
TypeScript

import { describe, expect, it } from 'vitest'
import { extractLabelFacts } from '../labels/label-schema.js'
import type { LifecycleEvent } from '../lifecycle/lifecycle-v0.js'
import type { GiteaIssue } from '../gitea/types.js'
import {
CALIBRATION_BUCKET_FLOOR,
calibrationCoverage,
calibrationSamples,
type CalibrationSample,
COLD_START_THRESHOLD,
fitCalibration,
toDurationModel,
} from './calibration-v0.js'
function sample(over: Partial<CalibrationSample> = {}): CalibrationSample {
return { issue: 1, estimateDays: 2, actualWorkingDays: 2, bucket: 2, person: null, ...over }
}
describe('fitCalibration', () => {
it('is cold-start below the threshold and reports the honest n', () => {
const m = fitCalibration([sample(), sample({ actualWorkingDays: 4 })])
expect(m.n).toBe(2)
expect(m.coldStart).toBe(true)
})
it('flips off cold-start at the threshold', () => {
const many = Array.from({ length: COLD_START_THRESHOLD }, (_, i) =>
sample({ issue: i, estimateDays: 2, actualWorkingDays: 3, bucket: 2 }),
)
const m = fitCalibration(many)
expect(m.n).toBe(COLD_START_THRESHOLD)
expect(m.coldStart).toBe(false)
})
it('recovers the global median ratio (mu = mean log-ratio)', () => {
// every actual is exactly 2x its estimate → mu = ln 2
const m = fitCalibration(Array.from({ length: 25 }, (_, i) => sample({ issue: i, estimateDays: 2, actualWorkingDays: 4 })))
expect(m.global.mu).toBeCloseTo(Math.log(2), 6)
})
it('fits a bucket only once it clears the floor', () => {
const twos = Array.from({ length: CALIBRATION_BUCKET_FLOOR }, (_, i) =>
sample({ issue: i, estimateDays: 2, actualWorkingDays: 3, bucket: 2 }),
)
const oneThin = [sample({ issue: 99, estimateDays: 5, actualWorkingDays: 9, bucket: 5 })]
const m = fitCalibration([...twos, ...oneThin])
expect(m.byBucket[2]?.n).toBe(CALIBRATION_BUCKET_FLOOR)
expect(m.byBucket[5]).toBeUndefined() // only 1 sample, below floor
})
it('drops non-positive estimates/actuals', () => {
const m = fitCalibration([sample({ actualWorkingDays: 0 }), sample({ estimateDays: 0 }), sample()])
expect(m.n).toBe(1)
})
it('derives a per-person bias relative to global', () => {
// one person consistently runs longer than the mean
const base = Array.from({ length: 20 }, (_, i) => sample({ issue: i, actualWorkingDays: 2, person: 'ak' }))
const slow = Array.from({ length: 3 }, (_, i) => sample({ issue: 100 + i, actualWorkingDays: 6, person: 'sm' }))
const m = fitCalibration([...base, ...slow])
expect(m.byPerson['sm'].biasMu).toBeGreaterThan(0)
expect(m.byPerson['ak'].biasMu).toBeLessThan(0)
})
})
describe('toDurationModel', () => {
it('projects the fit down to the params forecast needs', () => {
const m = fitCalibration(
Array.from({ length: 25 }, (_, i) => sample({ issue: i, estimateDays: 2, actualWorkingDays: 3, bucket: 2 })),
)
const dm = toDurationModel(m)
expect(dm.coldStart).toBe(false)
expect(dm.byBucket[2].mu).toBeCloseTo(m.byBucket[2].mu, 6)
expect(dm.global.mu).toBeCloseTo(m.global.mu, 6)
})
})
describe('calibrationSamples', () => {
const asOf = new Date('2026-02-01T00:00:00Z')
function issue(over: Partial<GiteaIssue>): GiteaIssue {
const labels = over.labels ?? []
return {
number: 1,
title: '#1',
body: '',
state: 'closed',
labels,
facts: extractLabelFacts(labels),
milestone: null,
assignee: null,
assignees: [],
createdAt: '2026-01-05T09:00:00Z',
updatedAt: '2026-01-12T09:00:00Z',
closedAt: '2026-01-12T09:00:00Z',
url: '',
...over,
}
}
const events = (i: number): Record<number, LifecycleEvent[]> => ({
[i]: [{ type: 'commit', at: '2026-01-07T09:00:00Z' }, { type: 'close', at: '2026-01-12T09:00:00Z' }],
})
it('samples closed, estimated issues with a resolvable actual', () => {
const i = issue({ number: 7, labels: ['est/2d'], assignee: 'sm', closedAt: '2026-01-12T09:00:00Z' })
const [s] = calibrationSamples([i], events(7), asOf)
expect(s.issue).toBe(7)
expect(s.estimateDays).toBe(2)
expect(s.bucket).toBe(2)
expect(s.person).toBe('sm')
// Wed 2026-01-07 → Mon 2026-01-12 = Wed,Thu,Fri = 3 working days
expect(s.actualWorkingDays).toBe(3)
})
it('skips open issues and closed ones without an estimate', () => {
const open = issue({ number: 8, state: 'open', labels: ['est/2d'], closedAt: null })
const noEst = issue({ number: 9, labels: [] })
expect(calibrationSamples([open, noEst], { ...events(8), ...events(9) }, asOf)).toEqual([])
})
it('coverage counts same-day closes as excluded candidates, not as "more closes needed"', () => {
// usable: commit Wed 01-07 → close Mon 01-12 = 3 working days
const usable = issue({ number: 7, labels: ['est/2d'] })
// same-day close: commit and close on the same day = 0 working days → excluded
const sameDay = issue({ number: 10, labels: ['est/2d'], createdAt: '2026-01-12T08:00:00Z' })
const sameDayEvents = {
10: [
{ type: 'commit', at: '2026-01-12T09:00:00Z' } as LifecycleEvent,
{ type: 'close', at: '2026-01-12T17:00:00Z' } as LifecycleEvent,
],
}
const open = issue({ number: 8, state: 'open', labels: ['est/2d'], closedAt: null })
const noEst = issue({ number: 9, labels: [] })
const cov = calibrationCoverage([usable, sameDay, open, noEst], { ...events(7), ...sameDayEvents }, asOf)
expect(cov.candidates).toBe(2) // closed + estimated only (usable + sameDay)
expect(cov.usable).toBe(1)
expect(cov.excludedSameDay).toBe(1)
// the honest denominator: usable matches the model's n
expect(cov.usable).toBe(calibrationSamples([usable, sameDay, open, noEst], { ...events(7), ...sameDayEvents }, asOf).length)
})
})