Compare commits

..

5 Commits

Author SHA1 Message Date
Croissant Le Doux
f08c4935dc Performance pass: benchmark the deterministic compute path (#32)
Lock in the PLAN.md compute targets so a regression that slips an O(n²) into the
scheduler or forecast fails the suite:
- scheduler + capacity layout + Monte Carlo forecast < 1s @ 200 open issues —
  measured 232ms, comfortable headroom.
- scaling stays ~linear (400 issues ≈ 3.9x the 100-issue time; asserts < 8x to
  rule out O(n²) while tolerating jitter).

Representative fixture: 200 open issues with varied estimates/priorities/assignees
across 3 capacity lanes + a light acyclic dependency web. Bounds are the real
targets with margin so timing jitter can't flake CI; actuals are logged.

Reconcile-<5s@500 is network-bound (~2N gitea calls) and stays covered by the live
reconcile — this benchmarks the pure compute the app runs each turn. +2 core tests.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-09 15:41:56 -04:00
2a6413821a Merge pull request 'calibration: count same-day closes honestly (#34)' (#54) from feat/calibration-honesty into main
Reviewed-on: #54
2026-07-09 19:23:59 +00:00
008435f1c2 Merge branch 'main' into feat/calibration-honesty 2026-07-09 19:23:55 +00:00
354ba9227e Merge pull request 'apply_changes: unified mutation tool — estimate/priority/assign/milestone, in-app + agent (#24)' (#53) from feat/apply-changes-assign-milestone into main
Reviewed-on: #53
2026-07-09 19:23:49 +00:00
Croissant Le Doux
89c873b368 calibration: count same-day closes honestly (#34)
The cold-start surface showed "N/20 closed issues estimated", implying you're
just (20−N) closes away. But calibrationSamples silently drops closed+estimated
issues that closed in 0 working days (same-day closes) — real closes that
structurally can't calibrate. On this repo that's 10 of 24 closes hidden: the
note read 14/20 as if 6 away, when a third of the history will never count.

- core: `calibrationCoverage(issues, timelines, asOf)` → { candidates, usable,
  excludedSameDay }, counting the silently-excluded same-day closes. Pure, tested.
- surface it: CalibrationData gains `excludedSameDay`; backlogCalibration returns
  the coverage; the Runway note and the Calibration screen now say "… · N same-day
  closes can't calibrate" so the thin sample is explained, not just reported.

Verified on christian/commitea: closed=24, usable=14, excludedSameDay=10.
131 core green (incl. new coverage test); core + desktop typecheck; 14 fixture e2e.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-09 15:18:16 -04:00
9 changed files with 163 additions and 5 deletions

View File

@@ -178,6 +178,9 @@ export function CalibrationScreen({ onBack, data }: { onBack: () => void; data?:
{c.active
? 'You are not bad at estimating; you are optimistic in a very stable way. Stable, I can work with.'
: 'Not enough closed history yet — Im forecasting from cold-start priors and widening the cone to stay honest. The curve takes over at 20.'}
{!c.active && c.excludedSameDay > 0
? ` And ${c.excludedSameDay} closed ${c.excludedSameDay === 1 ? 'issue' : 'issues'} closed the same day they were started — 0 working days cant calibrate, so they dont count toward the 20.`
: ''}
</p>
</Card>
</div>

View File

@@ -17,7 +17,7 @@ export function RunwayScreen({
}: {
onOpenCalibration: () => void
onOpenMilestone: (id?: number) => void
calibration?: { n: number; coldStart: boolean }
calibration?: { n: number; coldStart: boolean; excludedSameDay?: number }
milestones?: RunwayMilestone[]
capacity?: CapacityMember[]
}) {
@@ -30,9 +30,11 @@ export function RunwayScreen({
hours: `${capacityPerWorkday(m).toFixed(2)} pd/day`,
}))
: CAPACITY
const excluded = calibration?.excludedSameDay ?? 0
const calibNote = calibration
? calibration.coldStart
? `cold-start priors · ${calibration.n}/20 closed issues estimated`
? `cold-start priors · ${calibration.n}/20 closed issues estimated` +
(excluded > 0 ? ` · ${excluded} same-day close${excluded === 1 ? '' : 's'} cant calibrate` : '')
: `calibrated on ${calibration.n} closed ${calibration.n === 1 ? 'issue' : 'issues'}`
: 'calibrated on 27 closed issues'
return (

View File

@@ -306,7 +306,15 @@ export function AppShell() {
setMilestoneId(id ?? null)
setView('milestone')
}}
calibration={calibration ? { n: calibration.model.n, coldStart: calibration.model.coldStart } : undefined}
calibration={
calibration
? {
n: calibration.model.n,
coldStart: calibration.model.coldStart,
excludedSameDay: calibration.coverage.excludedSameDay,
}
: undefined
}
milestones={runwayMilestones}
capacity={capacityMembers}
/>

View File

@@ -361,11 +361,14 @@ export interface CalibrationData {
scatter: number[][]
fit: number
effect: { raw: string; banded: string; p50: string }
/** Closed+estimated issues that can't calibrate (same-day / 0-day closes). */
excludedSameDay: number
}
export const CALIBRATION: CalibrationData = {
n: 27,
active: true,
excludedSameDay: 0,
labels: [
{ label: 'est/1d', n: 8, median: '1.1d', bias: 8 },
{ label: 'est/2d', n: 9, median: '2.4d', bias: 18 },

View File

@@ -1,6 +1,8 @@
import {
type CalibrationCoverage,
type CalibrationModel,
type CalibrationSample,
calibrationCoverage,
calibrationSamples,
type CapacityMember,
capacityPerWorkday,
@@ -171,10 +173,11 @@ export function backlogCalibration(
issues: GiteaIssue[],
timelines: Timelines = {},
asOf: Date = new Date(),
): { model: CalibrationModel; data: CalibrationData } {
): { model: CalibrationModel; data: CalibrationData; coverage: CalibrationCoverage } {
const samples = calibrationSamples(issues, timelines, asOf)
const model = fitCalibration(samples)
return { model, data: calibrationData(model, samples, issues) }
const coverage = calibrationCoverage(issues, timelines, asOf)
return { model, coverage, data: calibrationData(model, samples, issues, coverage.excludedSameDay) }
}
const pctFromMu = (mu: number) => Math.round((Math.exp(mu) - 1) * 100)
@@ -188,6 +191,7 @@ export function calibrationData(
model: CalibrationModel,
samples: CalibrationSample[],
openIssues: GiteaIssue[],
excludedSameDay = 0,
): CalibrationData {
const labels = PRIOR_BUCKETS.map((b) => {
const inBucket = samples.filter((s) => s.bucket === b)
@@ -227,6 +231,7 @@ export function calibrationData(
scatter: samples.map((s) => [s.estimateDays, s.actualWorkingDays]),
fit: Number(Math.exp(model.global.mu).toFixed(2)),
effect,
excludedSameDay,
}
}

View File

@@ -5,6 +5,7 @@ import type { LifecycleEvent } from '../lifecycle/lifecycle-v0.js'
import type { GiteaIssue } from '../gitea/types.js'
import {
CALIBRATION_BUCKET_FLOOR,
calibrationCoverage,
calibrationSamples,
type CalibrationSample,
COLD_START_THRESHOLD,
@@ -118,4 +119,26 @@ describe('calibrationSamples', () => {
const noEst = issue({ number: 9, labels: [] })
expect(calibrationSamples([open, noEst], { ...events(8), ...events(9) }, asOf)).toEqual([])
})
it('coverage counts same-day closes as excluded candidates, not as "more closes needed"', () => {
// usable: commit Wed 01-07 → close Mon 01-12 = 3 working days
const usable = issue({ number: 7, labels: ['est/2d'] })
// same-day close: commit and close on the same day = 0 working days → excluded
const sameDay = issue({ number: 10, labels: ['est/2d'], createdAt: '2026-01-12T08:00:00Z' })
const sameDayEvents = {
10: [
{ type: 'commit', at: '2026-01-12T09:00:00Z' } as LifecycleEvent,
{ type: 'close', at: '2026-01-12T17:00:00Z' } as LifecycleEvent,
],
}
const open = issue({ number: 8, state: 'open', labels: ['est/2d'], closedAt: null })
const noEst = issue({ number: 9, labels: [] })
const cov = calibrationCoverage([usable, sameDay, open, noEst], { ...events(7), ...sameDayEvents }, asOf)
expect(cov.candidates).toBe(2) // closed + estimated only (usable + sameDay)
expect(cov.usable).toBe(1)
expect(cov.excludedSameDay).toBe(1)
// the honest denominator: usable matches the model's n
expect(cov.usable).toBe(calibrationSamples([usable, sameDay, open, noEst], { ...events(7), ...sameDayEvents }, asOf).length)
})
})

View File

@@ -125,3 +125,40 @@ export function calibrationSamples(
}
return out
}
/** How the closed+estimated backlog splits into usable samples vs. what can't calibrate. */
export interface CalibrationCoverage {
/** Closed issues carrying an estimate — the calibration candidates. */
candidates: number
/** Candidates that yielded a usable actual (> 0 working days) → become samples. */
usable: number
/**
* Candidates excluded because the issue closed with 0 working days (same-day
* close) or no resolvable actual — real closes that structurally can't
* calibrate. Counting them keeps `usable/threshold` honest: it's not "N more
* closes away" if some of your closes will never count.
*/
excludedSameDay: number
}
/**
* Coverage of the calibration candidates — how many closed+estimated issues are
* usable vs. silently unusable (same-day / 0-day closes). {@link calibrationSamples}
* drops the latter; this counts them so the UI can say *why* the sample is thin.
*/
export function calibrationCoverage(
issues: GiteaIssue[],
timelines: Record<number, LifecycleEvent[]>,
asOf: Date,
): CalibrationCoverage {
let candidates = 0
let usable = 0
for (const issue of issues) {
if (issue.state !== 'closed') continue
if (issue.facts.estimateDays == null) continue
candidates++
const inf = inferLifecycle(issue, timelines[issue.number] ?? [], asOf)
if (inf.actualWorkingDays != null && inf.actualWorkingDays > 0) usable++
}
return { candidates, usable, excludedSameDay: candidates - usable }
}

View File

@@ -80,6 +80,7 @@ export type {
export {
CALIBRATION_BUCKET_FLOOR,
calibrationCoverage,
calibrationSamples,
COLD_START_THRESHOLD,
fitCalibration,
@@ -87,6 +88,7 @@ export {
} from './calibration/calibration-v0.js'
export type {
BucketFit,
CalibrationCoverage,
CalibrationModel,
CalibrationSample,
PersonBias,

View File

@@ -0,0 +1,75 @@
/**
* Performance pass (#32). The deterministic compute path must stay well under the
* PLAN.md targets on representative fixtures:
* - scheduler + Monte Carlo forecast < 1s @ 200 open issues.
* - scaling stays roughly linear (no accidental O(n²) in the hot path).
*
* Reconcile-<5s@500 is network-bound (~2N gitea calls) and is covered by the live
* reconcile, not here — this file benchmarks the pure compute the app runs each
* turn. Bounds are the actual targets with comfortable headroom so timing jitter
* can't flake the suite; actuals are logged.
*/
import { describe, expect, it } from 'vitest'
import { forecast } from '../forecast/forecast-v0.js'
import { type DependencyEdge, schedule, type SchedulableIssue } from '../scheduler/scheduler-v0.js'
import { scheduleWithCapacity, type Worker } from '../scheduler/scheduler-capacity-v0.js'
const EST = [1, 2, 3, 5, 8]
const WORKERS: Worker[] = [
{ person: 'a', speed: 0.8 },
{ person: 'b', speed: 0.6 },
{ person: 'c', speed: 1.0 },
]
/** A representative open backlog: varied estimates/priorities/assignees + a light dependency web. */
function backlog(n: number): { issues: SchedulableIssue[]; edges: DependencyEdge[] } {
const issues: SchedulableIssue[] = Array.from({ length: n }, (_, i) => ({
number: i + 1,
title: `Issue ${i + 1} with a representative title of some length`,
labels: [`est/${EST[i % EST.length]}d`, `p/${(i % 4) + 1}`],
estimateDays: EST[i % EST.length],
priority: (i % 4) + 1,
assignee: WORKERS[i % WORKERS.length].person,
}))
// ~1 dependency per 3 issues, always on a lower-numbered issue (acyclic)
const edges: DependencyEdge[] = []
for (let i = 3; i < n; i += 3) edges.push({ issue: i + 1, dependsOn: i - 1 })
return { issues, edges }
}
function ms(fn: () => void): number {
const t0 = performance.now()
fn()
return performance.now() - t0
}
describe('perf (#32)', () => {
it('scheduler + Monte Carlo forecast < 1s @ 200 open issues', () => {
const { issues, edges } = backlog(200)
const elapsed = ms(() => {
schedule(issues, edges)
scheduleWithCapacity(issues, edges, WORKERS)
forecast(issues, edges, { workers: WORKERS }) // 2000 trials (default)
})
// eslint-disable-next-line no-console
console.log(`[perf] schedule+capacity+forecast @200 = ${elapsed.toFixed(1)}ms`)
expect(elapsed).toBeLessThan(1000)
})
it('scales roughly linearly — 400 issues is well under 4x the 100-issue time', () => {
const small = backlog(100)
const big = backlog(400)
const run = (b: typeof small) => () => {
schedule(b.issues, b.edges)
forecast(b.issues, b.edges, { workers: WORKERS })
}
// warm up (JIT) so the ratio reflects steady state
run(small)()
const t100 = Math.max(ms(run(small)), 0.1)
const t400 = ms(run(big))
// eslint-disable-next-line no-console
console.log(`[perf] @100 = ${t100.toFixed(1)}ms · @400 = ${t400.toFixed(1)}ms · ratio ${(t400 / t100).toFixed(1)}x`)
expect(t400).toBeLessThan(t100 * 8) // generous: rules out O(n²), tolerant of jitter
})
})