Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -216,3 +216,6 @@ __marimo__/

# Streamlit
.streamlit/secrets.toml

# NICE Data
data/nice/
8 changes: 8 additions & 0 deletions data/evals/ng28_retrieval.jsonl
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{"question": "What is the recommended frequency for measuring HbA1c levels in adults with type 2 diabetes once both the HbA1c level and blood glucose lowering therapy are stable?", "question_type": "verbatim", "answer": "6 months", "supporting_spans": ["6 months once the HbA1c level and blood glucose lowering therapy are stable."], "is_answerable": true, "notes": "", "_source_file": "data/nice/ng28/04_Blood-glucose-management.txt", "_eval_type": "retrieval", "_model": "google/gemma-4-31b-it:free", "_generated_at": "2026-06-19T00:04:32.889147+00:00"}
{"question": "What specific standardisation method should be used when measuring HbA1c levels in adults with type 2 diabetes?", "question_type": "verbatim", "answer": "International Federation of Clinical Chemistry (IFCC) standardisation", "supporting_spans": ["Measure HbA1c using methods calibrated according to International Federation of Clinical Chemistry (IFCC) standardisation."], "is_answerable": true, "notes": "", "_source_file": "data/nice/ng28/04_Blood-glucose-management.txt", "_eval_type": "retrieval", "_model": "google/gemma-4-31b-it:free", "_generated_at": "2026-06-19T00:04:32.889147+00:00"}
{"question": "For an adult with type 2 diabetes whose diabetes is managed solely through healthy living and diet, what is the target HbA1c level they should aim for?", "question_type": "paragraph", "answer": "They should aim for an HbA1c level of 48 mmol/mol (6.5%).", "supporting_spans": ["For adults whose type 2 diabetes is managed either by healthy living and diet, or healthy living and diet combined with an initial medication regimen that is not associated with hypoglycaemia (see the section on initial medicines ), support them to aim for an HbA1c level of 48 mmol/mol (6.5%)."], "is_answerable": true, "notes": "", "_source_file": "data/nice/ng28/04_Blood-glucose-management.txt", "_eval_type": "retrieval", "_model": "google/gemma-4-31b-it:free", "_generated_at": "2026-06-19T00:04:32.889147+00:00"}
{"question": "Under what circumstances should a clinician consider short-term self-monitoring of capillary blood glucose levels for an adult with type 2 diabetes?", "question_type": "paragraph", "answer": "Short-term self-monitoring should be considered when the patient is starting treatment with intravenous or oral corticosteroids, or to confirm suspected hypoglycaemia.", "supporting_spans": ["Consider short-term self-monitoring of capillary blood glucose levels in adults with type 2 diabetes, reviewing treatment as necessary:\n- when starting treatment with oral or intravenous corticosteroids or\n- to confirm suspected hypoglycaemia."], "is_answerable": true, "notes": "", "_source_file": "data/nice/ng28/04_Blood-glucose-management.txt", "_eval_type": "retrieval", "_model": "google/gemma-4-31b-it:free", "_generated_at": "2026-06-19T00:04:32.889147+00:00"}
{"question": "For an adult with type 2 diabetes on multiple daily insulin injections, what are the specific criteria for offering intermittently scanned continuous glucose monitoring (isCGM), and why is capillary blood glucose measurement still necessary for these patients?", "question_type": "multi_paragraph", "answer": "isCGM should be offered if the patient has recurrent or severe hypoglycaemia, impaired hypoglycaemia awareness, a condition or disability preventing capillary self-monitoring but allowing isCGM use, or if they would otherwise need to self-measure at least 8 times a day. Capillary measurements remain necessary to check the accuracy of the CGM device and to serve as a back-up (e.g., if the device stops working or blood glucose levels change quickly).", "supporting_spans": ["Offer intermittently scanned continuous glucose monitoring (isCGM, commonly referred to as 'flash') to adults with type 2 diabetes on multiple daily insulin injections if any of the following apply:\n- they have recurrent hypoglycaemia or severe hypoglycaemia\n- they have impaired hypoglycaemia awareness\n- they have a condition or disability (including a learning disability or cognitive impairment) that means they cannot self-monitor their blood glucose by capillary blood glucose monitoring but could use an isCGM device (or have it scanned for them)\n- they would otherwise be advised to self-measure at least 8 times a day.", "Advise adults with type 2 diabetes who are using CGM that they will still need to take capillary blood glucose measurements (although they can do this less often). Explain that is because:\n- they will need to use capillary blood glucose measurements to check the accuracy of their CGM device\n- they will need capillary blood glucose monitoring as a back-up (for example when their blood glucose levels are changing quickly or if the device stops working)."], "is_answerable": true, "notes": "", "_source_file": "data/nice/ng28/04_Blood-glucose-management.txt", "_eval_type": "retrieval", "_model": "google/gemma-4-31b-it:free", "_generated_at": "2026-06-19T00:04:32.889147+00:00"}
{"question": "If an adult with type 2 diabetes has an HbA1c level of 60 mmol/mol, what three actions should be taken according to the guidelines?", "question_type": "multi_paragraph", "answer": "The clinician should: 1) reinforce advice regarding adherence to medicines, healthy living, and diet; 2) support the person to aim for an HbA1c level of 53 mmol/mol (7.0%); and 3) intensify the patient's medicines.", "supporting_spans": ["In adults with type 2 diabetes, if HbA1c levels are not adequately controlled by the initial medication regimen and rise to 58 mmol/mol (7.5%) or higher:\n- reinforce advice about diet, healthy living and adherence to medicines and\n- support the person to aim for an HbA1c level of 53 mmol/mol (7.0%) and\n- intensify medicines."], "is_answerable": true, "notes": "", "_source_file": "data/nice/ng28/04_Blood-glucose-management.txt", "_eval_type": "retrieval", "_model": "google/gemma-4-31b-it:free", "_generated_at": "2026-06-19T00:04:32.889147+00:00"}
{"question": "Which validated scoring systems should be used in primary care to assess impaired hypoglycaemic awareness in adults with type 2 diabetes?", "question_type": "adversarial", "answer": "", "supporting_spans": [], "is_answerable": false, "notes": "The text mentions GOLD and Clarke scores but explicitly states that validated methods are 'not always available in primary care' and the committee did NOT recommend specific methods for assessment.", "_source_file": "data/nice/ng28/04_Blood-glucose-management.txt", "_eval_type": "retrieval", "_model": "google/gemma-4-31b-it:free", "_generated_at": "2026-06-19T00:04:32.889147+00:00"}
{"question": "What is the recommended annual frequency for the structured assessment of self-monitoring skills for patients on insulin?", "question_type": "adversarial", "answer": "", "supporting_spans": [], "is_answerable": false, "notes": "The document states a structured assessment should be carried out 'at least annually' for those self-monitoring, but it does not specify a different or specific frequency exclusively for those on insulin.", "_source_file": "data/nice/ng28/04_Blood-glucose-management.txt", "_eval_type": "retrieval", "_model": "google/gemma-4-31b-it:free", "_generated_at": "2026-06-19T00:04:32.889147+00:00"}
23 changes: 23 additions & 0 deletions datasets/amfv_datasets/eval_prompts/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
from amfv_datasets.eval_prompts.decomposition import (
DecompositionItem,
parse_response as parse_decomposition_response,
user_prompt as decomposition_user_prompt,
SYSTEM_PROMPT as DECOMPOSITION_SYSTEM_PROMPT,
)
from amfv_datasets.eval_prompts.retrieval import (
RetrievalItem,
parse_response as parse_retrieval_response,
user_prompt as retrieval_user_prompt,
SYSTEM_PROMPT as RETRIEVAL_SYSTEM_PROMPT,
)

__all__ = [
"DECOMPOSITION_SYSTEM_PROMPT",
"RETRIEVAL_SYSTEM_PROMPT",
"DecompositionItem",
"RetrievalItem",
"decomposition_user_prompt",
"parse_decomposition_response",
"parse_retrieval_response",
"retrieval_user_prompt",
]
204 changes: 204 additions & 0 deletions datasets/amfv_datasets/eval_prompts/decomposition.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,204 @@
from __future__ import annotations

import json
import re
from dataclasses import dataclass, field
from typing import Literal

ClaimType = Literal["factual", "hedged", "negation", "numeric", "procedural"]
SourceKind = Literal["model_output", "reasoning_trace", "document_passage", "multiple_choice_rationale"]


@dataclass
class DecompositionItem:
claim: str
claim_type: ClaimType
source_span: str
requires_coreference: bool = False
original_pronoun: str | None = None
resolved_referent: str | None = None
is_distractor: bool = False
distractor_reason: str | None = None
notes: str = ""


SYSTEM_PROMPT = """\
You are an expert medical claim decomposer building a reference evaluation \
dataset for a medical fact verification system.

Your task is to decompose a piece of medical text into a list of atomic, \
independently-verifiable claims, following the three Baichuan-M3 rules and the \
hard-case handling rules below.

CORE DECOMPOSITION RULES (Baichuan-M3)
=======================================
Rule 1 — Atomic claims with full coreference resolution
Each claim must be self-contained: anyone reading the claim alone, without \
the source text, must be able to look it up and verify it. Resolve all \
pronouns and anaphors before writing the claim.
Example: "She was started on lisinopril 10 mg daily." →
"The patient (45-year-old female with hypertension) was started on \
lisinopril 10 mg daily."

Rule 2 — Distractor filtering
When the source is a multiple-choice rationale, the author frequently recites \
incorrect options before refuting them. Do NOT decompose these distractor spans \
into claims. Mark them with is_distractor: true so human reviewers can verify \
the filtering. Example: "Option B, which states that amoxicillin is the first-\
line treatment for MRSA infections, is incorrect." → mark is_distractor: true \
with distractor_reason: "recitation of incorrect MC option".

Rule 3 — Deduplication with order preservation
If two spans assert the same fact (even with different phrasing), produce ONE \
claim and cite both source spans. Preserve the logical order of the original text.

HARD-CASE HANDLING RULES
=========================
Numeric precision
Never round, truncate, or paraphrase numbers, units, or lab values. Preserve \
them exactly. "eGFR 45 mL/min/1.73m²" must appear verbatim in the claim, not \
as "low eGFR" or "reduced kidney function".

Hedged claims
Preserve uncertainty markers. "may suggest", "is consistent with", "no \
definitive evidence" must appear in the claim text. Label these hedged.

Negated findings
"No evidence of pneumonia" is a verifiable claim. Do not drop negations or \
rephrase them as positive statements. Label these negation.

Long pronoun chains
When a pronoun references a subject defined more than two sentences earlier, \
set requires_coreference: true, record the pronoun in original_pronoun, and \
write the full referent in resolved_referent.

CLAIM TYPES
===========
factual — direct assertion of a medical fact.
hedged — source qualifies with uncertainty language.
negation — asserts absence or non-occurrence.
numeric — truth depends on a specific number, dose, lab value, or unit.
procedural — a step or ordered clinical action.

OUTPUT FORMAT
=============
Respond with a JSON array and nothing else—no preamble, no markdown fences, \
no trailing commentary. Each element must be an object with:

claim (string) Self-contained, coreference-resolved claim.
claim_type (string) One of: factual, hedged, negation, numeric, procedural
source_span (string) Exact substring(s) from the source; for multi-span,
join with " [...] ".
requires_coreference (boolean) true if pronoun resolution was needed.
original_pronoun (string|null)
resolved_referent (string|null)
is_distractor (boolean) true if this span should be filtered out.
distractor_reason (string|null)
notes (string) "" if none.

QUALITY RULES
=============
- Every non-distractor claim must be independently verifiable without the source.
- Do not merge distinct facts into one claim; split them if needed.
- Do not split a single atomic fact across multiple claims.
- Keep claims in the order they appear in the source text.
- A claim about a drug-dose-indication triple must include all three elements \
(e.g. "metformin 500 mg twice daily for type 2 diabetes mellitus").
"""


def user_prompt(
text: str,
*,
source_kind: SourceKind = "model_output",
document_title: str = "",
document_source: str = "",
extra_context: str = "",
) -> str:
kind_label = {
"model_output": "model output (final answer)",
"reasoning_trace": "model reasoning trace (chain-of-thought)",
"document_passage": "medical document passage",
"multiple_choice_rationale": "multiple-choice answer rationale",
}[source_kind]

header_parts = [f"Source kind: {kind_label}"]
if document_title:
header_parts.append(f"Title / question stem: {document_title}")
if document_source:
header_parts.append(f"Source: {document_source}")
if extra_context:
header_parts.append(f"Additional context: {extra_context}")
header = "\n".join(header_parts)

distractor_reminder = (
"\nNOTE: This text contains multiple-choice distractors. Apply Rule 2 "
"carefully—mark recitations of wrong options as is_distractor: true.\n"
if source_kind == "multiple_choice_rationale"
else ""
)

return f"""\
Decompose the following medical text into atomic, independently-verifiable claims.

{header}
{distractor_reminder}
TEXT
====
{text}
"""


_JSON_BLOCK_RE = re.compile(r"```(?:json)?\s*(.*?)\s*```", re.DOTALL)


def parse_response(raw: str) -> list[DecompositionItem]:
text = raw.strip()
match = _JSON_BLOCK_RE.search(text)
if match:
text = match.group(1)

try:
data = json.loads(text)
except json.JSONDecodeError as exc:
raise ValueError(f"Response is not valid JSON: {exc}\n\nRaw:\n{raw[:500]}") from exc

if not isinstance(data, list):
raise ValueError(f"Expected a JSON array at the top level, got {type(data).__name__}")

items: list[DecompositionItem] = []
for i, obj in enumerate(data):
try:
items.append(
DecompositionItem(
claim=obj["claim"],
claim_type=obj["claim_type"],
source_span=obj["source_span"],
requires_coreference=obj.get("requires_coreference", False),
original_pronoun=obj.get("original_pronoun"),
resolved_referent=obj.get("resolved_referent"),
is_distractor=obj.get("is_distractor", False),
distractor_reason=obj.get("distractor_reason"),
notes=obj.get("notes", ""),
)
)
except (KeyError, TypeError) as exc:
raise ValueError(f"Item {i} is missing required field: {exc}\n\nItem: {obj}") from exc

return items


def active_claims(items: list[DecompositionItem]) -> list[DecompositionItem]:
return [c for c in items if not c.is_distractor]


def distractor_claims(items: list[DecompositionItem]) -> list[DecompositionItem]:
return [c for c in items if c.is_distractor]


def claims_by_type(items: list[DecompositionItem], claim_type: ClaimType) -> list[DecompositionItem]:
return [c for c in items if c.claim_type == claim_type]


decomposition_user_prompt = user_prompt
parse_decomposition_response = parse_response
Loading