feat: the two readiness cards

Your score is the share of questions right at your most recent answer to
each. It is deliberately not called an equated score: AMBOSS's EPC rests
on psychometrics we do not have, and a number dressed up as one would be
a claim we cannot support. The card says so.

Against everyone else compares you with other learners on the questions
you have in common — not with their scores on whatever they happened to
sit. A percentile over different question sets reads someone who worked
through the hardest fifty in the bank as weaker than someone who did
fifty easy ones, which is the opposite of true.

Neither appears before it means anything, and each says which half is
missing: more questions of your own, more questions shared with others,
or more learners. The cohort reported is the most any one shared question
saw — distinct learners cannot be summed across questions without
counting the same person once per question.

The "readiness is still locked" note sat above the tab switch and so
appeared on Performance, where it described a table that is on the other
tab. Moved down to the table it is about.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01TqXevQJhxFrM7jJg82cgZN
This commit is contained in:
Daniel 2026-09-12 03:08:09 +02:00
parent bfc5ec93d7
commit 8a1b518502
6 changed files with 373 additions and 18 deletions

View file

@ -422,6 +422,116 @@ def performance_over_time(
}
#: Below this many other learners on the same questions there is no cohort to
#: compare against, only a couple of strangers.
PEER_MIN_LEARNERS = 3
#: And below this many questions held in common, the comparison is about which
#: questions each of you happened to sit.
PEER_MIN_SHARED = 10
def _latest_answers(db: Session, user_id: int):
"""The most recent answer to each question this learner has answered.
Repetitions, course quizzes, expired attempts and blanks are left out, as
everywhere that measures rather than counts practice. The most recent
answer is the one that says what is known now; the earlier attempt at the
same question says what was known then, which is a different question and
not the one a readiness score is asking.
"""
rows = db.query(
AttemptAnswer.question_id, AttemptAnswer.is_correct, QuizAttempt.completed_at,
).join(QuizAttempt, QuizAttempt.id == AttemptAnswer.attempt_id
).join(Quiz, Quiz.id == QuizAttempt.quiz_id
).filter(
QuizAttempt.user_id == user_id,
QuizAttempt.completed_at.isnot(None),
or_(QuizAttempt.expired == 0, QuizAttempt.expired.is_(None)),
Quiz.course_id.is_(None),
or_(Quiz.is_repetition == 0, Quiz.is_repetition.is_(None)),
AttemptAnswer.user_answer != "",
).order_by(QuizAttempt.completed_at.asc(), QuizAttempt.id.asc()).all()
latest: dict[int, bool] = {}
for row in rows:
latest[row.question_id] = bool(row.is_correct)
return latest
@router.get("/readiness")
def readiness(
db: Session = Depends(get_db),
user: User = Depends(get_current_user),
):
"""Two figures, each refusing to appear before it means anything.
**Your score** is the share of questions you got right at your most recent
attempt at each. Not an equated score: we do not have the psychometrics to
equate one, and a number dressed up as one would be a claim we cannot
support.
**Against everyone else** compares that with how other learners did on the
very questions you answered, rather than with their scores on whatever they
happened to sit. Someone who worked through the hardest fifty in the bank
should not read as weaker than someone who did fifty easy ones, and a
percentile over different question sets says exactly that.
"""
mine = _latest_answers(db, user.id)
answered = len(mine)
correct = sum(1 for right in mine.values() if right)
unlocked = answered >= READINESS_UNLOCK_ANSWERS
peers = {}
if mine:
rows = db.query(
AttemptAnswer.question_id,
func.count(func.distinct(QuizAttempt.user_id)).label("learners"),
func.count(AttemptAnswer.id).label("answers"),
func.sum(case((AttemptAnswer.is_correct.is_(True), 1), else_=0)).label("correct"),
).join(QuizAttempt, QuizAttempt.id == AttemptAnswer.attempt_id
).join(Quiz, Quiz.id == QuizAttempt.quiz_id
).filter(
QuizAttempt.user_id != user.id,
QuizAttempt.completed_at.isnot(None),
or_(QuizAttempt.expired == 0, QuizAttempt.expired.is_(None)),
Quiz.course_id.is_(None),
or_(Quiz.is_repetition == 0, Quiz.is_repetition.is_(None)),
AttemptAnswer.user_answer != "",
AttemptAnswer.question_id.in_(list(mine)),
).group_by(AttemptAnswer.question_id).all()
peers = {row.question_id: row for row in rows}
shared = [qid for qid in mine if qid in peers]
# Distinct learners cannot be summed across questions without counting the
# same person once per question they answered, so the cohort reported is
# the most any one shared question saw — a floor, not a guess.
cohort_size = max((peers[qid].learners for qid in shared), default=0)
peer_unlocked = len(shared) >= PEER_MIN_SHARED and cohort_size >= PEER_MIN_LEARNERS
expected = None
yours_on_shared = None
if shared:
expected = round(100 * sum(
(peers[qid].correct or 0) / peers[qid].answers for qid in shared) / len(shared), 1)
yours_on_shared = round(100 * sum(1 for qid in shared if mine[qid]) / len(shared), 1)
return {
"answered": answered,
"score": round(100 * correct / answered, 1) if answered else None,
"unlocked": unlocked,
"answers_needed": max(0, READINESS_UNLOCK_ANSWERS - answered),
"peer": {
"unlocked": peer_unlocked,
"shared_questions": len(shared),
"shared_needed": max(0, PEER_MIN_SHARED - len(shared)),
"cohort": cohort_size,
"cohort_needed": max(0, PEER_MIN_LEARNERS - cohort_size),
"expected": expected,
"yours": yours_on_shared,
"delta": None if expected is None else round(yours_on_shared - expected, 1),
},
}
@router.get("/recommendations")
def study_recommendations(
group: Literal["articles", "disciplines", "systems"] = "disciplines",

View file

@ -455,3 +455,79 @@ class PerformanceOverTimeTests(CompletionTests):
self.assertEqual(data['points'], [])
self.assertFalse(data['unlocked'])
self.assertEqual(data['total_answered'], 0)
class ReadinessTests(CompletionTests):
"""Two figures that refuse to appear before they mean anything."""
def read(self):
response = self.client.get('/study-tools/readiness')
self.assertEqual(response.status_code, 200, response.text)
return response.json()
def peer_sat(self, user_id, qid, correct, quiz_id=1):
from datetime import datetime
from app.models.attempt import AttemptAnswer, QuizAttempt
from app.models.user import User
if not self.bank.db.get(User, user_id):
self.bank.db.add(User(id=user_id, name=f'Learner {user_id}',
email=f'learner{user_id}@example.test',
hashed_password='unused'))
self.bank.db.flush()
attempt = QuizAttempt(user_id=user_id, quiz_id=quiz_id, total_questions=1,
score=int(correct), completed_at=datetime.utcnow())
self.bank.db.add(attempt)
self.bank.db.flush()
self.bank.db.add(AttemptAnswer(attempt_id=attempt.id, question_id=qid,
is_correct=correct, user_answer='yes', seconds_spent=30))
self.bank.db.commit()
def test_the_score_is_the_most_recent_answer_to_each_question(self):
self.sat(1, False, ago_days=9, seconds=30)
self.sat(1, True, ago_days=1, seconds=30)
self.sat(2, False, ago_days=1, seconds=30)
data = self.read()
# Two questions known about, one of them right: 50%, not the 33% a
# count of every answer ever given would report.
self.assertEqual(data['answered'], 2)
self.assertEqual(data['score'], 50.0)
def test_a_score_on_four_answers_is_not_shown_as_a_score(self):
for qid in (1, 2, 3):
self.sat(qid, True, ago_days=1, seconds=30)
data = self.read()
self.assertFalse(data['unlocked'])
self.assertEqual(data['answers_needed'], 37)
# The figure is still computed; the page decides whether to show it.
self.assertEqual(data['score'], 100.0)
def test_the_comparison_is_against_the_same_questions_not_the_same_people(self):
# Everyone answers question 1; the learner gets it right and two of the
# three peers get it wrong.
self.sat(1, True, ago_days=1, seconds=30)
self.peer_sat(2, 1, False)
self.peer_sat(3, 1, False)
self.peer_sat(4, 1, True)
peer = self.read()['peer']
self.assertEqual(peer['shared_questions'], 1)
self.assertEqual(peer['cohort'], 3)
self.assertEqual(peer['expected'], 33.3)
self.assertEqual(peer['yours'], 100.0)
self.assertEqual(peer['delta'], 66.7)
def test_no_cohort_means_no_comparison_and_it_says_what_is_missing(self):
self.sat(1, True, ago_days=1, seconds=30)
data = self.read()['peer']
self.assertFalse(data['unlocked'])
self.assertEqual(data['shared_questions'], 0)
self.assertEqual(data['cohort_needed'], 3)
self.assertIsNone(data['expected'])
self.assertIsNone(data['delta'])
def test_nothing_answered_reports_nothing(self):
data = self.read()
self.assertEqual(data['answered'], 0)
self.assertIsNone(data['score'])
self.assertFalse(data['unlocked'])
self.assertFalse(data['peer']['unlocked'])

View file

@ -308,16 +308,19 @@ Analysis**, which has three tabs.
### Performance
- [ ] **Readiness** — two cards. One is the headline score (AMBOSS calls it
EPC, an equated percent correct); the other is peer comparison. Both say
what is still needed before they mean anything — "complete 32 more
questions" — rather than showing a number built on four answers.
- [ ] **Next step: adaptive session** — a card with a question-count select and
one button. It repeats on the Recommendations tab; it is the thing the
whole page is for.
- [ ] **Performance over time** — a line of the headline score by date, with an
honest empty state ("complete more questions to unlock this chart")
rather than a chart drawn through two points.
- [x] **Readiness** — done 2026-09-12. `GET /study-tools/readiness`. Your
score is the most recent answer to each question, and it is *not* called
an equated score: we have no psychometrics to equate one. The peer card
compares you with other learners **on the questions you have in common**,
not with their scores on whatever they happened to sit — sitting the
hardest fifty should not read as weakness. Both stay dashes until they
mean something and say which of the two things they are waiting for.
- [x] **Next step: adaptive session** — already sat above the tab switch, so
it shows on both. The readiness-locked note did too, and has been moved
down to the table it is actually about.
- [x] **Performance over time** — done 2026-09-12. The running score across
everything answered up to each day, not the session on its own. Locked
below 40 answers or 3 sessions, saying which.
- [x] **Analysis panel beside it** — done 2026-09-12. `GET
/study-tools/answer-split` counts the same answers twice: every answer
ever given, and the most recent answer to each question. The donut is

View file

@ -175,3 +175,26 @@
}
.an-trend-now { font-size: 1.1rem; font-weight: 700; }
.an-trend .an-basis { margin-top: 10px; }
/* Readiness
Two cards, each showing a dash and its own reason rather than a number
built on four answers. */
.an-readiness {
display: grid; grid-template-columns: minmax(0, 1fr) minmax(0, 1fr);
gap: 18px; margin-top: 4px;
}
.an-ready {
padding: 16px 18px 18px;
border: 1px solid var(--border); border-radius: 12px; background: var(--card-bg);
}
.an-ready h2 { margin: 0 0 8px; font-size: 1.05rem; font-weight: 650; }
.an-ready-figure { margin: 0 0 8px; font-size: 2.1rem; font-weight: 700; line-height: 1.1; }
.an-ready-figure.is-locked { color: var(--text-subtle); }
.an-ready-figure.is-ahead { color: var(--correct-fg); }
.an-ready-figure.is-behind { color: var(--wrong-fg); }
.an-ready-unit { font-size: 0.8rem; font-weight: 600; color: var(--text-muted); margin-left: 6px; }
.an-ready-note { margin: 0; font-size: 0.78rem; line-height: 1.6; color: var(--text-muted); }
@media (max-width: 900px) {
.an-readiness { grid-template-columns: minmax(0, 1fr); }
}

View file

@ -94,6 +94,92 @@ function Completion() {
)
}
/**
* The two figures a learner opens this page for.
*
* Neither appears before it means anything, and each says what it is waiting
* for. A score built on four answers is not a small score, it is not a score,
* and showing one because the arithmetic is possible is how a page starts
* lying politely.
*/
function Readiness() {
const [data, setData] = useState(null)
const [error, setError] = useState('')
useEffect(() => {
let live = true
api.get('/study-tools/readiness')
.then(res => { if (live) setData(res.data) })
.catch(() => { if (live) setError('Could not load your readiness') })
return () => { live = false }
}, [])
if (error) return <p className="an-empty" role="alert">{error}</p>
if (!data) return <p className="an-empty">Loading</p>
const peer = data.peer || {}
const ahead = peer.delta > 0
const level = Math.abs(peer.delta ?? 0) < 2
return (
<div className="an-readiness">
<section className="an-card an-ready">
<h2>Your score</h2>
{data.unlocked ? (
<>
<p className="an-ready-figure">{data.score}%</p>
<p className="an-ready-note">
Of the {data.answered} questions you have answered, counting your
most recent answer to each. Not an equated score we do not have
the psychometrics to equate one, and a number dressed up as one
would be a claim we cannot support.
</p>
</>
) : (
<>
<p className="an-ready-figure is-locked"></p>
<p className="an-ready-note">
Answer {data.answers_needed} more question
{data.answers_needed === 1 ? '' : 's'} and this becomes a score.
Below that it would be arithmetic, not a measurement.
</p>
</>
)}
</section>
<section className="an-card an-ready">
<h2>Against everyone else</h2>
{peer.unlocked ? (
<>
<p className={`an-ready-figure ${ahead ? 'is-ahead' : level ? '' : 'is-behind'}`}>
{level ? 'Level' : `${ahead ? '+' : ''}${peer.delta}`}
{!level && <span className="an-ready-unit">points</span>}
</p>
<p className="an-ready-note">
You scored {peer.yours}% on the {peer.shared_questions} questions
you have in common with other learners; they averaged{' '}
{peer.expected}%. Compared on the same questions rather than on
whatever each person happened to sit working through the hardest
fifty in the bank should not read as weakness.
</p>
</>
) : (
<>
<p className="an-ready-figure is-locked"></p>
<p className="an-ready-note">
{peer.shared_needed > 0
? `Answer ${peer.shared_needed} more question${peer.shared_needed === 1 ? '' : 's'} that other learners have also sat.`
: `Waiting for ${peer.cohort_needed} more learner${peer.cohort_needed === 1 ? '' : 's'} to sit the questions you have.`}
{' '}Until then there is nobody to compare with, only a couple of
strangers.
</p>
</>
)}
</section>
</div>
)
}
/**
* Whether the score is going anywhere.
*
@ -369,14 +455,6 @@ export default function AnalysisPage() {
</div>
</div>
{!data.unlocked && (
<div className="an-locked">
<strong>Readiness is still locked</strong>
Answer {data.answers_needed} more question{data.answers_needed !== 1 ? 's' : ''} and each topic gets a
readiness estimate. Until then the table ranks topics by raw accuracy and how much of the bank they cover.
</div>
)}
<div className="an-nextstep">
<div className="an-nextstep-text">
<h2>Next step: adaptive session</h2>
@ -395,6 +473,16 @@ export default function AnalysisPage() {
{tab === 'recommendations' ? (
<>
{/* About the table below, so it sits with the table rather than
on a tab that does not have one. */}
{!data.unlocked && (
<div className="an-locked">
<strong>Readiness is still locked</strong>
Answer {data.answers_needed} more question{data.answers_needed !== 1 ? 's' : ''} and each topic gets a
readiness estimate. Until then the table ranks topics by raw accuracy and how much of the bank they cover.
</div>
)}
{/* The same answers asked three ways: which reading to go back
to, which discipline is weak, which organ system is weak. */}
<div className="an-group-toggle" role="tablist" aria-label="Group knowledge profile by">
@ -435,6 +523,7 @@ export default function AnalysisPage() {
</>
) : (
<>
<Readiness />
<OverTime />
<div className="an-perf-top">
<Completion />

View file

@ -192,3 +192,57 @@ it('will not draw a trend it does not have, and says which half is missing', asy
await userEvent.click(tabs[tabs.length - 1])
expect(await screen.findByText('Complete 1 more session to unlock this chart.')).toBeInTheDocument()
})
const ready = (over = {}) => ({
answered: 60, score: 72.0, unlocked: true, answers_needed: 0,
peer: {
unlocked: true, shared_questions: 40, shared_needed: 0, cohort: 12,
cohort_needed: 0, expected: 64.0, yours: 72.0, delta: 8.0,
},
...over,
})
const withReadiness = (data) => {
api.get.mockImplementation(url =>
Promise.resolve({ data: url === '/study-tools/readiness' ? data : payload() }))
}
const performanceTab = async () => {
render(<MemoryRouter><AnalysisPage /></MemoryRouter>)
const tabs = await screen.findAllByRole('tab', { name: 'Performance' })
await userEvent.click(tabs[tabs.length - 1])
}
it('shows the score and how it compares on the same questions', async () => {
withReadiness(ready())
await performanceTab()
const card = (await screen.findByText('Your score')).closest('.an-ready')
expect(within(card).getByText('72%')).toBeInTheDocument()
expect(within(card).getByText(/most recent answer to each/)).toBeInTheDocument()
const peer = screen.getByText('Against everyone else').closest('.an-ready')
expect(within(peer).getByText('+8')).toBeInTheDocument()
expect(within(peer).getByText(/they averaged 64%/)).toBeInTheDocument()
})
it('will not show a score built on four answers', async () => {
withReadiness(ready({ answered: 8, score: 100.0, unlocked: false, answers_needed: 32 }))
await performanceTab()
const card = (await screen.findByText('Your score')).closest('.an-ready')
expect(within(card).getByText('—')).toBeInTheDocument()
expect(within(card).queryByText('100%')).not.toBeInTheDocument()
expect(within(card).getByText(/Answer 32 more questions/)).toBeInTheDocument()
})
it('says whether the comparison is short of questions or short of people', async () => {
withReadiness(ready({ peer: { unlocked: false, shared_questions: 4, shared_needed: 6,
cohort: 5, cohort_needed: 0, expected: null, yours: null, delta: null } }))
await performanceTab()
expect(await screen.findByText(/Answer 6 more questions that other learners have also sat/))
.toBeInTheDocument()
withReadiness(ready({ peer: { unlocked: false, shared_questions: 40, shared_needed: 0,
cohort: 1, cohort_needed: 2, expected: null, yours: null, delta: null } }))
await performanceTab()
expect(await screen.findByText(/Waiting for 2 more learners/)).toBeInTheDocument()
})