From c8fd9d77b124f1a8704c528418b5cf2442a0d9f9 Mon Sep 17 00:00:00 2001 From: Maira Salazar Date: Wed, 2 Sep 2026 19:52:30 +0200 Subject: [PATCH 1/3] feat(schema): match and normalize for comparison pair Add `matches()`/`normalize_for_comparison()` to Creator, ResolvedLicense and ResolvedFunding so `_align()` can pair suggested vs. current metadata fields (from the deposit form). Since `normalize_for_comparison()` requires the funder's name from the id, we also move FUNDER_ROR_IDS to extracted_metadata.py and derive FUNDER_NAMES_BY_ROR_ID from it. --- app/activities/resolve_metadata.py | 15 +----------- app/schemas/extracted_metadata.py | 18 ++++++++++++++ app/schemas/metadata_suggestions.py | 16 ++++++++++++- app/schemas/resolved_fields.py | 37 +++++++++++++++++++++++++++++ 4 files changed, 71 insertions(+), 15 deletions(-) diff --git a/app/activities/resolve_metadata.py b/app/activities/resolve_metadata.py index 9a292ac..bb94b2b 100644 --- a/app/activities/resolve_metadata.py +++ b/app/activities/resolve_metadata.py @@ -16,7 +16,7 @@ from app.activities.utils import http_verify from app.config import get_settings -from app.schemas.extracted_metadata import ExtractedMetadata, FunderEnum +from app.schemas.extracted_metadata import FUNDER_ROR_IDS, ExtractedMetadata, FunderEnum from app.schemas.metadata_suggestions import ( FundingSuggestion, LicenseSuggestion, @@ -40,19 +40,6 @@ _INVENIO_HEADERS = {"Accept": "application/vnd.inveniordm.v1+json"} -FUNDER_ROR_IDS = { - FunderEnum.NIH: "01cwqze88", - FunderEnum.NSF: "021nxhr62", - FunderEnum.UKRI: "001aqnf71", - FunderEnum.FNS: "00yjd3n13", - FunderEnum.EC: "00k4n6c32", - FunderEnum.FCT: "00snfqn58", - FunderEnum.NWO: "04jsz6e67", - FunderEnum.NHMRC: "011kf5r70", - FunderEnum.ANR: "00rbzpz17", - FunderEnum.ARC: "05mmh0f86", -} - class ResolveMetadataRequest(BaseModel): """Request to resolve and generate metadata suggestions from raw metadata.""" diff --git a/app/schemas/extracted_metadata.py b/app/schemas/extracted_metadata.py index a58c0dc..30b929d 100644 --- a/app/schemas/extracted_metadata.py +++ b/app/schemas/extracted_metadata.py @@ -26,6 +26,24 @@ class FunderEnum(str, Enum): ARC = "Australian Research Council" +FUNDER_ROR_IDS = { + FunderEnum.NIH: "01cwqze88", + FunderEnum.NSF: "021nxhr62", + FunderEnum.UKRI: "001aqnf71", + FunderEnum.FNS: "00yjd3n13", + FunderEnum.EC: "00k4n6c32", + FunderEnum.FCT: "00snfqn58", + FunderEnum.NWO: "04jsz6e67", + FunderEnum.NHMRC: "011kf5r70", + FunderEnum.ANR: "00rbzpz17", + FunderEnum.ARC: "05mmh0f86", +} + +FUNDER_NAMES_BY_ROR_ID = { + ror_id: funder.value for funder, ror_id in FUNDER_ROR_IDS.items() +} + + class ExtractedMetadata(BaseModel): """Flat schema the LLM fills, converted to ``MetadataSuggestions``. diff --git a/app/schemas/metadata_suggestions.py b/app/schemas/metadata_suggestions.py index 320b428..4a36175 100644 --- a/app/schemas/metadata_suggestions.py +++ b/app/schemas/metadata_suggestions.py @@ -3,8 +3,9 @@ """Typed metadata suggestions returned by the workflow.""" -# from __future__ import annotations +from difflib import SequenceMatcher +# from __future__ import annotations from typing import Annotated, Literal from idutils.normalizers import normalize_orcid @@ -52,6 +53,19 @@ def normalize_name(cls, v: str) -> str: given = " ".join(parts[:-1]) return f"{family}, {given}" + def matches(self, other: Creator) -> bool: + """Logic to pair two creator entries.""" + name_self = " ".join(self.name.replace(",", "").split()).casefold() + name_other = " ".join(other.name.replace(",", "").split()).casefold() + return ( + name_self == name_other + or SequenceMatcher(None, name_self, name_other).ratio() >= 0.85 + ) + + def normalize_for_comparison(self, matched=None) -> dict: + """Return the dictionary with name, affiliation, and ORCID.""" + return {"name": self.name, "orcid": self.orcid, "affiliation": self.affiliation} + class TitleSuggestion(BaseModel): """Suggestion for `title`.""" diff --git a/app/schemas/resolved_fields.py b/app/schemas/resolved_fields.py index bd866ae..2b11ccc 100644 --- a/app/schemas/resolved_fields.py +++ b/app/schemas/resolved_fields.py @@ -3,8 +3,12 @@ """Resolved metadata fields.""" +from difflib import SequenceMatcher + from pydantic import BaseModel, Field +from app.schemas.extracted_metadata import FUNDER_NAMES_BY_ROR_ID + class ResolvedFunder(BaseModel): """A funding organization resolved against the funders vocabulary.""" @@ -40,6 +44,19 @@ class ResolvedFunding(BaseModel): funder: ResolvedFunder | None = None award: ResolvedAward | None = None + def normalize_for_comparison(self, matched=None) -> dict: + """Normalize for comparisons.""" + normalized = {} + if self.funder: + normalized["funder_id"] = self.funder.id + normalized["funder_name"] = self.funder.name or FUNDER_NAMES_BY_ROR_ID.get( + self.funder.id + ) + if self.award: + for k in self.award.model_dump().keys(): + normalized["award_" + k] = getattr(self.award, k) + return normalized + class ResolvedLicense(BaseModel): """A license resolved against the Invenio licenses vocabulary.""" @@ -57,3 +74,23 @@ class ResolvedLicense(BaseModel): "https://opensource.org/licenses/MIT", ], ) + + def matches(self, other: ResolvedLicense) -> bool: + """Logic to pair two license entries.""" + if self.id.casefold() == other.id.casefold(): + return True + if not self.title or not other.title: + return False + return ( + SequenceMatcher(None, self.title.casefold(), other.title.casefold()).ratio() + >= 0.8 + ) + + def normalize_for_comparison(self, matched=None) -> dict: + """Normalize for comparisons. Borrows the resolved license title on a match.""" + normalized = {"id": self.id} + if self.title: + normalized["title"] = self.title + elif matched and matched.title: + normalized["title"] = matched.title + return normalized From a40d17142296b8f3ce2bb07e269f4363e82d929f Mon Sep 17 00:00:00 2001 From: Maira Salazar Date: Wed, 2 Sep 2026 19:52:07 +0200 Subject: [PATCH 2/3] feat(workflow): add compare_metadata workflow Add the `compare_metadata_with_llm` activity to compare pairs of metadata fields (suggested vs current metadata). The LLM judges whether the current metadata describes the underlying file. The `CompareMetadata` workflow calls the PDF text extraction, LLM metadata suggestion, vocabulary resolution, and the new comparison activity. --- app/activities/__init__.py | 3 + app/activities/compare_metadata.py | 126 +++++++ app/schemas/metadata_comparison.py | 260 +++++++++++++++ app/workflows/compare_metadata_workflow.py | 142 ++++++++ app/workflows/registry.py | 10 + tests/test_compare_metadata.py | 363 +++++++++++++++++++++ 6 files changed, 904 insertions(+) create mode 100644 app/activities/compare_metadata.py create mode 100644 app/schemas/metadata_comparison.py create mode 100644 app/workflows/compare_metadata_workflow.py create mode 100644 tests/test_compare_metadata.py diff --git a/app/activities/__init__.py b/app/activities/__init__.py index 42de0a1..4b2c0ac 100644 --- a/app/activities/__init__.py +++ b/app/activities/__init__.py @@ -4,6 +4,7 @@ """Activities for the Orcha application.""" from .check_funding_relevance import check_funding_relevance +from .compare_metadata import compare_metadata_with_llm from .extract_metadata import extract_metadata_with_llm from .extract_pdf_content import extract_pdf_text from .resolve_metadata import resolve_metadata_suggestions @@ -15,6 +16,7 @@ resolve_metadata_suggestions, update_workflow, check_funding_relevance, + compare_metadata_with_llm, ] __all__ = [ @@ -24,4 +26,5 @@ "resolve_metadata_suggestions", "update_workflow", "check_funding_relevance", + "compare_metadata_with_llm", ] diff --git a/app/activities/compare_metadata.py b/app/activities/compare_metadata.py new file mode 100644 index 0000000..474c56c --- /dev/null +++ b/app/activities/compare_metadata.py @@ -0,0 +1,126 @@ +# SPDX-FileCopyrightText: 2026 CERN. +# SPDX-License-Identifier: MIT + +"""LLM-based metadata comparison activity.""" + +import json +from datetime import timedelta + +from pydantic import BaseModel, Field +from temporalio import activity +from temporalio.common import RetryPolicy + +from app.agent import build_agent +from app.config import get_settings +from app.observability import propagate_langfuse_context +from app.schemas.metadata_comparison import ( + LIST_FIELDS, + SCALAR_FIELDS, + ComparedMetadata, + CurrentMetadata, + MetadataComparisons, +) +from app.schemas.metadata_suggestions import MetadataSuggestions +from app.workflows.specs import WorkflowContext + +COMPARE_METADATA_RETRY_POLICY = RetryPolicy( + initial_interval=timedelta(seconds=5), + backoff_coefficient=2, + maximum_interval=timedelta(seconds=20), + maximum_attempts=3, +) + + +class CompareMetadataRequest(BaseModel): + """Request to compare two sets of metadata.""" + + suggested_metadata: MetadataSuggestions = Field( + description="Metadata suggestions generated from text" + ) + current_metadata: CurrentMetadata = Field( + description="Current fields from the deposit form" + ) + + +INSTRUCTIONS = ( + "Compare each already aligned pair of fields given to decide if they can be " + "considered to describe the same underlying document. Base every explanation " + "only on the values present in the pair you are given; When one side is null, " + "state plainly that the value is missing from that side. Give a reasoning per " + "pair and an overall decision on the document (for which values absent in " + "suggested but present in current metadata are acceptable)." +) + + +def _align(suggested_metadata: list, current_metadata: list) -> list[dict]: + """Pair two lists by matching items, or keeping unmatched items paired with None.""" + remaining = list(suggested_metadata) + pairs = [] + for current in current_metadata: + match = next( + (suggested for suggested in remaining if current.matches(suggested)), None + ) + if match is not None: + remaining.remove(match) + pairs.append( + { + "suggested": match.normalize_for_comparison() + if match is not None + else None, + "current": current.normalize_for_comparison(matched=match), + } + ) + return pairs + [ + {"suggested": suggested.normalize_for_comparison(), "current": None} + for suggested in remaining + ] + + +def _build_pairs(suggested: MetadataSuggestions, current: CurrentMetadata) -> str: + scalar_fields: dict[str, str] = {} + list_fields: dict[str, list] = {} + for s in suggested.suggestions: + if isinstance(s.value, list): + list_fields[s.field] = s.value + else: + scalar_fields[s.field] = s.value + + pairs = {} + for field in SCALAR_FIELDS: + suggested_field = scalar_fields.get(field) + current_field = getattr(current, field) + if suggested_field or current_field: + pairs[field] = { + "suggested": suggested_field or None, + "current": current_field or None, + } + for field in LIST_FIELDS: + suggested_field = list_fields.get(field, []) + current_field = getattr(current, field) + if suggested_field or current_field: + pairs[field] = _align(suggested_field, current_field) + + return json.dumps(pairs, indent=1, ensure_ascii=False) + + +@activity.defn +async def compare_metadata_with_llm( + request: CompareMetadataRequest, + context: WorkflowContext, +) -> MetadataComparisons: + """Compare two sets of metadata using an LLM.""" + suggested_metadata = request.suggested_metadata + if not suggested_metadata.suggestions: + return MetadataComparisons( + comparisons=[], + describes_file=False, + decision="File does not have enough text for the comparison.", + ) + current_metadata = request.current_metadata + pairs = _build_pairs(suggested_metadata, current_metadata) + + agent = build_agent(get_settings().llm, ComparedMetadata, INSTRUCTIONS) + with propagate_langfuse_context(context, trace_name="compare_metadata"): + result = await agent.run(pairs) + + return result.output.to_comparisons(json.loads(pairs)) diff --git a/app/schemas/metadata_comparison.py b/app/schemas/metadata_comparison.py new file mode 100644 index 0000000..d901369 --- /dev/null +++ b/app/schemas/metadata_comparison.py @@ -0,0 +1,260 @@ +# SPDX-FileCopyrightText: 2026 CERN. +# SPDX-License-Identifier: MIT + +"""Metadata comparison returned by the workflow.""" + +from difflib import SequenceMatcher +from typing import Literal + +from pydantic import BaseModel, Field + +from app.schemas.extracted_metadata import FUNDER_NAMES_BY_ROR_ID +from app.schemas.metadata_suggestions import Creator +from app.schemas.resolved_fields import ResolvedFunding, ResolvedLicense + +SCALAR_FIELDS = ("title", "description", "publication_date", "doi", "copyright") +LIST_FIELDS = ("creators", "license", "funding") + +##################### +# Current fields # +##################### + + +class CurrentFunder(BaseModel): + """Funder entry from deposit metadata.""" + + id: str | None = None + name: str | None = None + + +class CurrentAward(BaseModel): + """Award entry from deposit metadata.""" + + id: str | None = None + number: str | None = None + title: dict[str, str] | None = None # e.g. {"en": "..."} + + +class CurrentFunding(BaseModel): + """Funding entry from deposit metadata: funder and/or award.""" + + funder: CurrentFunder | None = None + award: CurrentAward | None = None + + def matches(self, other: ResolvedFunding) -> bool: + """Logic to pair two funding entries.""" + funder_matches = False + if self.funder and other.funder: + funder_matches = self.funder.id == other.funder.id + if not (self.award and other.award): + return funder_matches + + award_matches = False + if self.award and other.award: + if self.award.id and self.award.id == other.award.id: + award_matches = True + elif self.award.number == other.award.number: + award_matches = True + elif self.award.title and other.award.title: + award_en = self.award.title.get("en", "") + if ( + SequenceMatcher( + None, award_en.casefold(), other.award.title.casefold() + ).ratio() + >= 0.85 + ): + award_matches = True + + return funder_matches and award_matches + + def normalize_for_comparison(self, matched: ResolvedFunding | None = None) -> dict: + """Normalize for comparisons. Borrows the resolved award title on a match.""" + normalized = {} + if self.funder: + normalized["funder_id"] = self.funder.id + normalized["funder_name"] = self.funder.name or ( + FUNDER_NAMES_BY_ROR_ID.get(self.funder.id) if self.funder.id else None + ) + if self.award: + for k in self.award.model_dump().keys(): + normalized["award_" + k] = getattr(self.award, k) + # Title comes from the matched suggested award + if matched and matched.award and matched.award.title: + normalized["award_title"] = matched.award.title + # Multilanguage dict + elif isinstance(normalized.get("award_title"), dict): + title = normalized["award_title"] + normalized["award_title"] = title.get("en") or next( + iter(title.values()), None + ) + return normalized + + +class CurrentMetadata(BaseModel): + """Deposit metadata.""" + + title: str = "" + description: str = "" + creators: list[Creator] = Field(default_factory=list) + publication_date: str = "" + doi: str = "" + license: list[ResolvedLicense] = Field(default_factory=list) + copyright: str = "" + funding: list[CurrentFunding] = Field(default_factory=list) + + +##################### +# Comparison fields # +##################### + + +class MetadataComparison(BaseModel): + """Field comparison.""" + + field: Literal[ + "title", + "description", + "creators", + "doi", + "publication_date", + "license", + "copyright", + "funding", + ] + rationale: str + suggested: str | list[dict | None] | None = None + current: str | list[dict | None] | None = None + + +class MetadataComparisons(BaseModel): + """Container for all metadata comparisons from a workflow run.""" + + comparisons: list[MetadataComparison] + describes_file: bool + decision: str + + +class ComparedMetadata(BaseModel): + """Flat schema the LLM fills, converted to ``MetadataComparisons``.""" + + title: str | None = Field( + default=None, + description="Correctness of the field `Title`", + examples=[ + "The title contains a typo in the word 'Weather'", + "The suggested title and the current title are the same", + ], + ) + description: str | None = Field( + default=None, + description="Correctness of the field `Description`. Ignore html tags.", + examples=["The abstract paraphrases the original abstract"], + ) + creators: str | None = Field( + default=None, + description=( + "Correctness of the field `Creators`. Take into consideration the name, " + "ORCID and affiliation." + ), + examples=[ + "Missing author van der Berg, A.", + "Author Doe, Jane is missing ORCID 0000-0002-1111-1115", + "Affiliation does not match for author Doe, John", + ], + ) + doi: str | None = Field( + default=None, + description="Correctness of The Digital Object Identifier, as a bare DOI " + "without a URL prefix", + examples=[ + "DOI matches the one in the suggestions", + "Missing DOI.", + ], + ) + publication_date: str | None = Field( + default=None, + description=( + "Whether publication date matches suggested publication date. " + "If suggested date's precision is different from the current date's " + "precision, indicate that." + ), + examples=["Document does not indicate day, only 2014-07"], + ) + license: str | None = Field( + default=None, + description="Whether the current `License` matches the suggested, based on the " + "id and name.", + examples=["License `MIT` does not match license `Apache 2.0` in the text"], + ) + copyright: str | None = Field( + default=None, + description="Correctness of `Copyright` field.", + examples=["Copyright matches", "Missing year in copyright"], + ) + funding: str | None = Field( + default=None, + description=( + "Whether the current `funding` recorded on the deposit matches the funding " + "found in the document. Having the same funder (unless it is null) is a " + "necessary but not sufficient criteria for matching awards. " + ), + examples=[ + "The award matches the one found in the document", + "The award titles are similar, but the numbers do not match", + "The document has a European Commission grant but there is no current " + "award number in the deposit", + ], + ) + describes_file: bool = Field( + description="Whether the current fields describes the underlying file" + ) + decision: str = Field(description="Explanation of the describes_file decision") + + def to_comparisons(self, pairs: dict[str, dict]) -> MetadataComparisons: + """Build the typed comparisons, dropping null/empty fields.""" + + def get_clean_values( + entry: dict[str, str | None] | None, + ) -> dict[str, str] | None: + if not entry: + return None + result = {} + for key in entry.keys(): + if value := entry[key]: + result[key] = value + return result + + comparisons: list[MetadataComparison] = [] + for field in SCALAR_FIELDS: + if rationale := getattr(self, field): + pair = pairs[field] + comparisons.append( + MetadataComparison( + field=field, + rationale=rationale, + suggested=pair.get("suggested"), + current=pair.get("current"), + ) + ) + + for field in LIST_FIELDS: + if rationale := getattr(self, field): + field_list = pairs[field] + comparisons.append( + MetadataComparison( + field=field, + rationale=rationale, + suggested=[ + get_clean_values(x.get("suggested")) for x in field_list + ], + current=[ + get_clean_values(x.get("current")) for x in field_list + ], + ) + ) + + return MetadataComparisons( + comparisons=comparisons, + describes_file=self.describes_file, + decision=self.decision, + ) diff --git a/app/workflows/compare_metadata_workflow.py b/app/workflows/compare_metadata_workflow.py new file mode 100644 index 0000000..8ab443d --- /dev/null +++ b/app/workflows/compare_metadata_workflow.py @@ -0,0 +1,142 @@ +# SPDX-FileCopyrightText: 2026 CERN. +# SPDX-License-Identifier: MIT + +from datetime import timedelta + +from pydantic import Field, HttpUrl +from temporalio import workflow + +from app.activities import ( + extract_metadata_with_llm, + extract_pdf_text, + resolve_metadata_suggestions, +) +from app.activities.compare_metadata import ( + COMPARE_METADATA_RETRY_POLICY, + CompareMetadataRequest, + compare_metadata_with_llm, +) +from app.activities.extract_metadata import ( + EXTRACT_METADATA_RETRY_POLICY, + ExtractMetadataRequest, +) +from app.activities.extract_pdf_content import ( + EXTRACT_PDF_TEXT_RETRY_POLICY, + ExtractPdfContentRequest, +) +from app.activities.resolve_metadata import ( + RESOLVE_METADATA_RETRY_POLICY, + ResolveMetadataRequest, +) +from app.activities.update_workflow import ( + UPDATE_WORKFLOW_RETRY_POLICY, + WorkflowUpdateRequest, + update_workflow, +) +from app.database.models import WorkflowStatus +from app.schemas.metadata_comparison import CurrentMetadata, MetadataComparisons +from app.workflows.specs import WorkflowContext, WorkflowParams + + +class CompareMetadataParams(WorkflowParams): + """Params for the compare_metadata workflow.""" + + url: HttpUrl + extractor: str = "pdfplumber" + pages: list[int] | None = Field(default_factory=lambda: [1, 2]) + metadata: CurrentMetadata + + +@workflow.defn +class CompareMetadata: + """Workflow that checks if a record's metadata matches the content of the record.""" + + @workflow.run + async def run( + self, + context: WorkflowContext, + params: CompareMetadataParams, + ) -> MetadataComparisons: + """Execute the metadata comparison check.""" + try: + await workflow.execute_activity( + update_workflow, + WorkflowUpdateRequest( + public_id=context.workflow_id, + tenant_id=context.tenant_id, + start_time=workflow.now(), + ), + start_to_close_timeout=timedelta(minutes=1), + retry_policy=UPDATE_WORKFLOW_RETRY_POLICY, + ) + + # Activity 1: Extract PDF text + content = await workflow.execute_activity( + extract_pdf_text, + ExtractPdfContentRequest( + url=str(params.url), + extractor=params.extractor, + pages=params.pages, + ), + start_to_close_timeout=timedelta(minutes=5), + retry_policy=EXTRACT_PDF_TEXT_RETRY_POLICY, + ) + + # Activity 2: Generate raw metadata suggestions using LLM + metadata = await workflow.execute_activity( + extract_metadata_with_llm, + args=[ExtractMetadataRequest(text=content.text), context], + start_to_close_timeout=timedelta(minutes=5), + retry_policy=EXTRACT_METADATA_RETRY_POLICY, + ) + + # Activity 3: Resolve funders, awards, and licenses; format suggestions + suggestions = await workflow.execute_activity( + resolve_metadata_suggestions, + ResolveMetadataRequest(metadata=metadata), + start_to_close_timeout=timedelta(minutes=3), + retry_policy=RESOLVE_METADATA_RETRY_POLICY, + ) + + # Activity 4: compare the metadata suggestions with the deposit metadata + result = await workflow.execute_activity( + compare_metadata_with_llm, + args=[ + CompareMetadataRequest( + suggested_metadata=suggestions, current_metadata=params.metadata + ), + context, + ], + start_to_close_timeout=timedelta(minutes=5), + retry_policy=COMPARE_METADATA_RETRY_POLICY, + ) + + except Exception: + await workflow.execute_activity( + update_workflow, + WorkflowUpdateRequest( + public_id=context.workflow_id, + tenant_id=context.tenant_id, + status=WorkflowStatus.ERROR, + result=None, + end_time=workflow.now(), + ), + start_to_close_timeout=timedelta(minutes=1), + retry_policy=UPDATE_WORKFLOW_RETRY_POLICY, + ) + raise + + await workflow.execute_activity( + update_workflow, + WorkflowUpdateRequest( + public_id=context.workflow_id, + tenant_id=context.tenant_id, + status=WorkflowStatus.SUCCESS, + result=result.model_dump(), + end_time=workflow.now(), + ), + start_to_close_timeout=timedelta(minutes=1), + retry_policy=UPDATE_WORKFLOW_RETRY_POLICY, + ) + + return result diff --git a/app/workflows/registry.py b/app/workflows/registry.py index 93dcbc5..6f6dcbb 100644 --- a/app/workflows/registry.py +++ b/app/workflows/registry.py @@ -9,6 +9,10 @@ CheckFundingRelevance, CheckFundingRelevanceParams, ) +from app.workflows.compare_metadata_workflow import ( + CompareMetadata, + CompareMetadataParams, +) from app.workflows.extract_metadata_workflow import ( ExtractMetadata, ExtractMetadataParams, @@ -29,6 +33,12 @@ task_queue=DEFAULT_TASK_QUEUE, id_prefix="check-funding-relevance", ), + "compare_metadata": WorkflowSpec( + workflow_cls=CompareMetadata, + params_model=CompareMetadataParams, + task_queue=DEFAULT_TASK_QUEUE, + id_prefix="compare-metadata", + ), } diff --git a/tests/test_compare_metadata.py b/tests/test_compare_metadata.py new file mode 100644 index 0000000..849ba45 --- /dev/null +++ b/tests/test_compare_metadata.py @@ -0,0 +1,363 @@ +# SPDX-FileCopyrightText: 2026 CERN. +# SPDX-License-Identifier: MIT + +"""Tests for the compare_metadata activity.""" + +import asyncio +import json + +import pytest + +from app.activities.compare_metadata import ( + CompareMetadataRequest, + _align, + _build_pairs, + compare_metadata_with_llm, +) +from app.schemas.metadata_comparison import ( + ComparedMetadata, + CurrentAward, + CurrentFunder, + CurrentFunding, + CurrentMetadata, +) +from app.schemas.metadata_suggestions import ( + Creator, + CreatorsSuggestion, + DescriptionSuggestion, + DoiSuggestion, + MetadataSuggestions, + PublicationDateSuggestion, + TitleSuggestion, +) +from app.schemas.resolved_fields import ( + ResolvedAward, + ResolvedFunder, + ResolvedFunding, + ResolvedLicense, +) +from app.workflows.specs import WorkflowContext + +EC_ROR_ID = "00k4n6c32" +NIH_ROR_ID = "01cwqze88" +NSF_ROR_ID = "021nxhr62" + + +def test_align_no_match(): + """On no match, two pairs are created.""" + suggested = [Creator(name="Doe, Jane")] + current = [Creator(name="Poppins, Mary")] + + pairs = _align(suggested, current) + + assert pairs == [ + { + "suggested": None, + "current": { + "name": "Poppins, Mary", + "orcid": None, + "affiliation": None, + }, + }, + { + "suggested": { + "name": "Doe, Jane", + "orcid": None, + "affiliation": None, + }, + "current": None, + }, + ] + + +def test_align_none(): + """If one of the metadata sets does not have a field, pair is built with None.""" + suggested = [Creator(name="Doe, Jane")] + + assert _align(suggested, []) == [ + { + "suggested": { + "name": "Doe, Jane", + "orcid": None, + "affiliation": None, + }, + "current": None, + } + ] + + +def test_align_license_id(): + """Licenses match by id.""" + suggested = [ + ResolvedLicense( + id="cc-by-4.0", title="Creative Commons Attribution 4.0 International" + ) + ] + current = [ResolvedLicense(id="cc-by-4.0")] + + # Matches, so one pair is created, and the current title is filled in + assert _align(suggested, current) == [ + { + "suggested": { + "id": "cc-by-4.0", + "title": "Creative Commons Attribution 4.0 International", + }, + "current": { + "id": "cc-by-4.0", + "title": "Creative Commons Attribution 4.0 International", + }, + } + ] + + +@pytest.mark.parametrize( + "title", + [ + "Creative Commons Attribution 4.0 International", + "Creative Commons Attribution 4.0", + ], +) +def test_align_license_name(title): + """Licenses match by a (near-)identical title when the id differs.""" + suggested = [ + ResolvedLicense( + id="cc-by-4.0", title="Creative Commons Attribution 4.0 International" + ) + ] + current = [ResolvedLicense(id="", title=title)] + + # Matches, so one pair is created, and the current title is retained + assert _align(suggested, current) == [ + { + "suggested": { + "id": "cc-by-4.0", + "title": "Creative Commons Attribution 4.0 International", + }, + "current": { + "id": "", + "title": title, + }, + } + ] + + +def test_align_funder_name_on_match(): + """A funder with only an id gets its name filled in on a match.""" + suggested = [ + ResolvedFunding(funder=ResolvedFunder(id=EC_ROR_ID, name="European Commission")) + ] + current = [CurrentFunding(funder=CurrentFunder(id=EC_ROR_ID, name=None))] + + assert _align(suggested, current) == [ + { + "suggested": {"funder_id": EC_ROR_ID, "funder_name": "European Commission"}, + "current": {"funder_id": EC_ROR_ID, "funder_name": "European Commission"}, + } + ] + + +@pytest.mark.parametrize("title", [None, {"en": ""}, {"en": "Test"}, {"en": "SCOAP3"}]) +def test_align_award_title_on_match(title): + """A matched award gets the resolved award's title on match.""" + suggested = [ + ResolvedFunding( + funder=ResolvedFunder(id=EC_ROR_ID, name="European Commission"), + award=ResolvedAward( + id=f"{EC_ROR_ID}::101166718", number="101166718", title="SCOAP3" + ), + ) + ] + current = [ + CurrentFunding( + funder=CurrentFunder(id=EC_ROR_ID, name=None), + award=CurrentAward(number="101166718", title=title), + ) + ] + + pairs = _align(suggested, current) + + assert len(pairs) == 1 + assert pairs[0]["suggested"]["award_title"] == "SCOAP3" + # The multilingual dict is replaced by the plain resolved title on a match. + assert pairs[0]["current"]["award_title"] == "SCOAP3" + + +@pytest.mark.parametrize( + ("title_dict", "title"), + [ + ({"fr": "Projet", "en": "Project"}, "Project"), + ({"fr": "Projet", "es": "Projeto"}, "Projet"), + ], +) +def test_align_multilingual_title_without_match(title_dict, title): + """Without a match, the raw multilingual title is flattened. + + Uses 'en' if available, otherwise the first available language is used instead. + """ + current = [ + CurrentFunding( + funder=CurrentFunder(id=NIH_ROR_ID, name="National Institutes of Health"), + award=CurrentAward(number="123", title=title_dict), + ) + ] + + pairs = _align([], current) + + assert pairs == [ + { + "suggested": None, + "current": { + "funder_id": NIH_ROR_ID, + "funder_name": "National Institutes of Health", + "award_id": None, + "award_number": "123", + "award_title": title, + }, + } + ] + + +def test_build_pairs_scalar_fields(): + """Correctly build pairs for scalar fields. + + A field present in both sets builds a pair; if it's only present in one + set, builds a pair with None. + """ + suggested = MetadataSuggestions( + suggestions=[ + TitleSuggestion(value="Suggested Title"), + DoiSuggestion(value="10.1234/example.5678"), + ] + ) + current = CurrentMetadata(title="Current Title") + + pairs = json.loads(_build_pairs(suggested, current)) + + assert pairs == { + "title": {"suggested": "Suggested Title", "current": "Current Title"}, + "doi": {"suggested": "10.1234/example.5678", "current": None}, + } + + +def test_build_pairs_aligns_list_fields(): + """List fields (e.g. creators) go through `_align`, not the scalar branch.""" + suggested = MetadataSuggestions( + suggestions=[CreatorsSuggestion(value=[Creator(name="Doe, Jane")])] + ) + current = CurrentMetadata(creators=[Creator(name="Doe, Jane")]) + + pairs = json.loads(_build_pairs(suggested, current)) + + assert pairs["creators"] == [ + { + "suggested": {"name": "Doe, Jane", "orcid": None, "affiliation": None}, + "current": {"name": "Doe, Jane", "orcid": None, "affiliation": None}, + } + ] + + +def test_no_suggested_metadata(): + """No suggested metadata returns empty comparisons, no LLM call.""" + request = CompareMetadataRequest( + suggested_metadata=MetadataSuggestions(suggestions=[]), + current_metadata=CurrentMetadata(title="Current Title"), + ) + context = WorkflowContext(workflow_id="wf-1", tenant_id="t-1") + result = asyncio.run(compare_metadata_with_llm(request, context)) + + assert result.comparisons == [] + assert not result.describes_file + + +def test_to_comparisons(): + """MetadataComparisons is built correctly given a valid CompareMetadataRequest.""" + suggested_metadata = MetadataSuggestions( + suggestions=[ + TitleSuggestion(value="A title that matches"), + DescriptionSuggestion( + value="We discuss the theoretical bases that underpin matching." + ), + CreatorsSuggestion( + value=[ + Creator(name="Frederik, R.", affiliation="CERN"), + Creator(name="Doe, S.", affiliation="CERN"), + ] + ), + DoiSuggestion(value="10.1234/example.5678"), + PublicationDateSuggestion(value="2014-07-17"), + ] + ) + + current_metadata = CurrentMetadata( + title="A title that matcches", # contains a typo + description="We discuss the theoretical basis that underpin matching.", + creators=[Creator(name="Frederik, R")], + publication_date="2014-07", + doi="10.1234/example.5678", + license=[ResolvedLicense(id="cc-by-4.0")], + copyright="The Authors", + ) + + request = CompareMetadataRequest( + suggested_metadata=suggested_metadata, current_metadata=current_metadata + ) + pairs = _build_pairs(request.suggested_metadata, request.current_metadata) + + mocked_output = ComparedMetadata.model_validate( + { + "title": "The suggested title and the current title are the same except " + "for a typo.", + "description": "Both descriptions are essentially the same; minor wording " + "differences ('bases' vs 'basis') can be ignored.", + "creators": "Creator lists match; discrepancies are missing affiliation and" + " one author not present in current; acceptable.", + "doi": "DOI matches the one in the extraction.", + "publication_date": "Suggested date '2014-07-17' is more precise than " + "'2014-07' – the current date provides only year " + "and month, but refers to the same document.", + "license": "Suggested metadata missing license, current metadata specifies " + "cc‑by‑4.0; acceptable missing in suggested.", + "copyright": "Missing in suggested metadata, present as 'The Authors'.", + "funding": None, + "describes_file": True, + "decision": "All provided metadata components either match or exhibit " + "differences that are considered acceptable (missing fields in " + "suggested or current), indicating that the current metadata " + "describes the same underlying document.", + } + ) + + result = mocked_output.to_comparisons(json.loads(pairs)) + # Scalar field: Title + assert result.comparisons[0].model_dump() == { + "field": "title", + "rationale": "The suggested title and the current title are the same except " + "for a typo.", + "suggested": "A title that matches", + "current": "A title that matcches", + } + + # Scalar field present in only one metadata set: Copyright + assert result.comparisons[4].model_dump() == { + "field": "copyright", + "rationale": "Missing in suggested metadata, present as 'The Authors'.", + "suggested": None, + "current": "The Authors", + } + + # List field: Creators, with a None value in one of the metadata sets + assert result.comparisons[5].model_dump() == { + "field": "creators", + "rationale": "Creator lists match; discrepancies are missing affiliation and " + "one author not present in current; acceptable.", + "suggested": [ + {"name": "Frederik, R.", "affiliation": "CERN"}, + {"name": "Doe, S.", "affiliation": "CERN"}, + ], + "current": [{"name": "Frederik, R"}, None], + } + + # 7 fields, empty field `funding` is dropped + assert len(result.comparisons) == 7 + assert result.describes_file is True + assert result.decision From f4297d8cfff63e36b26c9832dcb40c63358d99ea Mon Sep 17 00:00:00 2001 From: Maira Salazar Date: Wed, 9 Sep 2026 15:13:21 +0200 Subject: [PATCH 3/3] refactor(tests): move activity tests to dedicated directory --- tests/{ => activities}/test_check_funding_relevance.py | 0 tests/{ => activities}/test_compare_metadata.py | 0 tests/{ => activities}/test_extract_metadata.py | 0 tests/{ => activities}/test_resolve_metadata.py | 0 4 files changed, 0 insertions(+), 0 deletions(-) rename tests/{ => activities}/test_check_funding_relevance.py (100%) rename tests/{ => activities}/test_compare_metadata.py (100%) rename tests/{ => activities}/test_extract_metadata.py (100%) rename tests/{ => activities}/test_resolve_metadata.py (100%) diff --git a/tests/test_check_funding_relevance.py b/tests/activities/test_check_funding_relevance.py similarity index 100% rename from tests/test_check_funding_relevance.py rename to tests/activities/test_check_funding_relevance.py diff --git a/tests/test_compare_metadata.py b/tests/activities/test_compare_metadata.py similarity index 100% rename from tests/test_compare_metadata.py rename to tests/activities/test_compare_metadata.py diff --git a/tests/test_extract_metadata.py b/tests/activities/test_extract_metadata.py similarity index 100% rename from tests/test_extract_metadata.py rename to tests/activities/test_extract_metadata.py diff --git a/tests/test_resolve_metadata.py b/tests/activities/test_resolve_metadata.py similarity index 100% rename from tests/test_resolve_metadata.py rename to tests/activities/test_resolve_metadata.py