From 8c72cf1b4f8797c1260e26079288c00570911f13 Mon Sep 17 00:00:00 2001 From: ta4tsering Date: Tue, 21 Oct 2025 12:33:34 +0530 Subject: [PATCH 01/15] refactor: deleted tests and modified files --- tests/pecha/test_annotation.py | 366 +---- tests/pecha/test_update.py | 26 - .../26E4/segmentation-Tm3Uewnh3ySsvgIE.json | 1320 ----------------- tests/test_ids.py | 19 - 4 files changed, 1 insertion(+), 1730 deletions(-) delete mode 100644 tests/pecha/test_update.py delete mode 100644 tests/pecha/update/data/ID8Sv2ynVKZX8wIt/layers/26E4/segmentation-Tm3Uewnh3ySsvgIE.json diff --git a/tests/pecha/test_annotation.py b/tests/pecha/test_annotation.py index 87868116..bf23821b 100644 --- a/tests/pecha/test_annotation.py +++ b/tests/pecha/test_annotation.py @@ -1,16 +1,10 @@ -import json - import pytest from pydantic import ValidationError from openpecha.pecha.annotations import ( - AnnotationModel, BaseAnnotation, - PechaAlignment, - PechaId, span, ) -from openpecha.pecha.layer import AnnotationType def test_span_end_must_not_be_less_than_start(): @@ -21,362 +15,4 @@ def test_span_end_must_not_be_less_than_start(): def test_annotation_id(): ann = BaseAnnotation(span=span(start=10, end=20)) assert ann.span.start == 10 - assert ann.span.end == 20 - assert ann.metadata is None - - -def test_pechaid_valid(): - pid = PechaId.validate("I1234ABCD") - assert pid == "I1234ABCD" - - -def test_pechaid_invalid(): - with pytest.raises(ValueError): - PechaId.validate("X1234ABCD") - with pytest.raises(ValueError): - PechaId.validate("I1234ABC") # too short - with pytest.raises(ValueError): - PechaId.validate("I1234ABCDE") # too long - with pytest.raises(ValueError): - PechaId.validate("I1234abcD") # lowercase - - -def test_pecha_alignment_fields(): - pa = PechaAlignment(pecha_id="I1234ABCD", alignment_id="align1") - assert pa.pecha_id == "I1234ABCD" - assert pa.alignment_id == "align1" - - -def test_annotation_model_minimal_alignment(): - align = PechaAlignment(pecha_id="I1234ABCD", alignment_id="align1") - am = AnnotationModel( - pecha_id="I1234ABCD", - type=AnnotationType.ALIGNMENT, - document_id="doc1", - path="ann1", - title="Test", - aligned_to=align, - ) - assert am.pecha_id == "I1234ABCD" - assert am.type == AnnotationType.ALIGNMENT - assert am.aligned_to == align - - -def test_annotation_model_minimal_non_alignment(): - am = AnnotationModel( - pecha_id="I1234ABCD", - type=AnnotationType.SEGMENTATION, - document_id="doc1", - path="ann1", - title="Test", - ) - assert am.pecha_id == "I1234ABCD" - assert am.type == AnnotationType.SEGMENTATION - assert am.aligned_to is None - - -def test_annotation_model_with_alignment(): - align = PechaAlignment(pecha_id="I1234ABCD", alignment_id="align1") - am = AnnotationModel( - pecha_id="I1234ABCD", - type=AnnotationType.ALIGNMENT, - document_id="doc1", - path="ann1", - title="Test", - aligned_to=align, - ) - assert am.aligned_to is not None - assert am.aligned_to.pecha_id == "I1234ABCD" - assert am.aligned_to.alignment_id == "align1" - - -def test_annotation_model_invalid_pechaid(): - with pytest.raises(ValidationError): - AnnotationModel( - pecha_id="BADID", - type=AnnotationType.ALIGNMENT, - document_id="doc1", - path="ann1", - title="Test", - ) - - -def test_annotation_model_missing_required(): - with pytest.raises(ValidationError): - AnnotationModel( - pecha_id="I1234ABCD", - type=AnnotationType.ALIGNMENT, - document_id="doc1", - # path missing - title="Test", - ) - - -class TestValidAnnotationModel: - """Tests for valid annotation models in different scenarios.""" - - def test_valid_annotation_minimal(self): - """Test minimal valid annotation with default values.""" - input_data = { - "pecha_id": "I12345678", - "document_id": "DOC123", - "title": "Test Annotation", - "path": "E11/layer.json", - } - - model = AnnotationModel(**input_data) - assert str(model.pecha_id) == "I12345678" - assert model.document_id == "DOC123" - assert model.title == "Test Annotation" - assert model.type == AnnotationType.SEGMENTATION - assert model.aligned_to is None - - def test_valid_annotation_with_type(self): - """Test valid annotation with explicit type.""" - input_data = { - "pecha_id": "I12345678", - "document_id": "DOC123", - "title": "Test Alignment Annotation", - "type": "alignment", - "path": "E11/layer.json", - } - - model = AnnotationModel(**input_data) - assert str(model.pecha_id) == "I12345678" - assert model.document_id == "DOC123" - assert model.title == "Test Alignment Annotation" - assert model.type == AnnotationType.ALIGNMENT - assert model.aligned_to is None - - def test_valid_annotation_with_alignment(self): - """Test valid annotation with alignment information.""" - input_data = { - "pecha_id": "I12345678", - "document_id": "DOC123", - "title": "Test Annotation with Alignment", - "type": "alignment", - "path": "E11/layer.json", - "aligned_to": { - "pecha_id": "I87654321", - "alignment_id": "ALIGN001", - }, - } - - model = AnnotationModel(**input_data) - assert str(model.pecha_id) == "I12345678" - assert model.document_id == "DOC123" - assert model.title == "Test Annotation with Alignment" - assert model.type == AnnotationType.ALIGNMENT - assert model.aligned_to is not None - assert model.model_dump()["aligned_to"]["pecha_id"] == "I87654321" - assert model.model_dump()["aligned_to"]["alignment_id"] == "ALIGN001" - - def test_valid_annotation_from_dict(self): - """Test creating a valid annotation from a dictionary.""" - input_data = { - "pecha_id": "I12345678", - "document_id": "DOC123", - "title": "Test Dict Annotation", - "path": "E11/layer.json", - } - - model = AnnotationModel.model_validate(input_data) - assert str(model.pecha_id) == "I12345678" - assert model.document_id == "DOC123" - assert model.title == "Test Dict Annotation" - assert model.type == AnnotationType.SEGMENTATION - - -class TestInvalidAnnotationModel: - """Tests for invalid annotation models that should raise validation errors.""" - - def test_invalid_pecha_id_format(self): - """Test that invalid pecha_id format raises ValidationError.""" - with pytest.raises(ValidationError) as exc_info: - AnnotationModel( - pecha_id="invalid_id", # Should start with I and contain 8 hex chars - document_id="DOC123", - title="Invalid ID Test", - path="E11/layer.json", - ) - - # Check the specific validation error message - errors = exc_info.value.errors() - assert any( - err["loc"] == ("pecha_id",) - and "PechaId must start with 'I' followed by 8 uppercase hex characters" - in err["msg"] - for err in errors - ) - - def test_missing_document_id(self): - """Test that missing document_id raises ValidationError.""" - with pytest.raises(ValidationError) as exc_info: - AnnotationModel( - pecha_id="I12345678", - # Missing document_id - title="Missing Document ID Test", - ) - - errors = exc_info.value.errors() - assert any("document_id" in str(err) for err in errors) - assert any("Field required" in str(err) for err in errors) - - def test_missing_title(self): - """Test that missing title raises ValidationError.""" - with pytest.raises(ValidationError) as exc_info: - AnnotationModel( - pecha_id="I12345678", - document_id="DOC123", - # Missing title - ) - - errors = exc_info.value.errors() - assert any("title" in str(err) for err in errors) - assert any("Field required" in str(err) for err in errors) - - def test_empty_title(self): - """Test that empty title raises ValidationError.""" - with pytest.raises(ValidationError) as exc_info: - AnnotationModel( - pecha_id="I12345678", - document_id="DOC123", - title="", # Empty title - ) - - errors = exc_info.value.errors() - assert any("title" in str(err) for err in errors) - assert any("min_length" in str(err) for err in errors) - - def test_invalid_document_id(self): - """Test that invalid document_id raises ValidationError.""" - with pytest.raises(ValidationError) as exc_info: - AnnotationModel( - pecha_id="I12345678", - document_id="", # Empty document_id - title="Invalid Document ID Test", - ) - - errors = exc_info.value.errors() - assert any("document_id" in str(err) for err in errors) - assert any("pattern" in str(err) for err in errors) - - def test_invalid_type(self): - """Test that invalid type raises ValidationError.""" - with pytest.raises(ValidationError) as exc_info: - AnnotationModel( - pecha_id="I12345678", - document_id="DOC123", - title="Invalid Type Test", - type="invalid_type", # Not in AnnotationType enum - ) - - errors = exc_info.value.errors() - assert any("type" in str(err) for err in errors) - assert any("enum" in str(err) for err in errors) - - def test_invalid_aligned_to(self): - """Test that invalid aligned_to raises ValidationError.""" - with pytest.raises(ValidationError) as exc_info: - AnnotationModel( - pecha_id="I12345678", - document_id="DOC123", - title="Invalid Alignment Test", - type="Alignment", - path="E11/layer.json", - aligned_to={ - "pecha_id": "invalid_id", # Invalid pecha_id format - "alignment_id": "ALIGN001", - }, - ) - - errors = exc_info.value.errors() - assert any("aligned_to" in str(err) for err in errors) - assert any( - err["loc"] == ("aligned_to", "pecha_id") - and "PechaId must start with 'I' followed by 8 uppercase hex characters" - in err["msg"] - for err in errors - ) - - def test_missing_alignment_id(self): - """Test that missing alignment_id in aligned_to raises ValidationError.""" - with pytest.raises(ValidationError) as exc_info: - AnnotationModel( - pecha_id="I12345678", - document_id="DOC123", - title="Missing Alignment ID Test", - type="alignment", - aligned_to={ - "pecha_id": "I87654321", - # Missing alignment_id - }, - ) - - errors = exc_info.value.errors() - assert any("aligned_to" in str(err) for err in errors) - - -class TestAnnotationModelSerialization: - """Tests for serialization of annotation models.""" - - def test_model_dump(self): - """Test model_dump() produces the expected dictionary.""" - model = AnnotationModel( - pecha_id="I12345678", - document_id="DOC123", - title="Serialization Test", - type="alignment", - path="E11/layer.json", - aligned_to={ - "pecha_id": "I87654321", - "alignment_id": "ALIGN001", - }, - ) - - data = model.model_dump() - assert data["pecha_id"] == "I12345678" # Should be string not nested object - assert data["document_id"] == "DOC123" - assert data["title"] == "Serialization Test" - assert data["type"] == AnnotationType.ALIGNMENT - assert data["path"] == "E11/layer.json" - assert data["aligned_to"]["pecha_id"] == "I87654321" - assert data["aligned_to"]["alignment_id"] == "ALIGN001" - - def test_model_dump_json(self): - """Test model_dump_json() produces valid JSON with expected structure.""" - model = AnnotationModel( - pecha_id="I12345678", - document_id="DOC123", - title="JSON Serialization Test", - path="E11/layer.json", - ) - - json_str = model.model_dump_json() - data = json.loads(json_str) - - assert data["pecha_id"] == "I12345678" - assert data["document_id"] == "DOC123" - assert data["title"] == "JSON Serialization Test" - assert data["type"] == "segmentation" - assert data["path"] == "E11/layer.json" - assert data["aligned_to"] is None - - def test_json_schema(self): - """Test the JSON schema is correctly generated.""" - schema = AnnotationModel.model_json_schema() - - # Check basic schema structure - assert "properties" in schema - assert "pecha_id" in schema["properties"] - assert "document_id" in schema["properties"] - assert "title" in schema["properties"] - assert "type" in schema["properties"] - assert "aligned_to" in schema["properties"] - - # Check required fields - assert "required" in schema - required_fields = schema["required"] - assert "pecha_id" in required_fields - assert "document_id" in required_fields - assert "title" in required_fields + assert ann.span.end == 20 \ No newline at end of file diff --git a/tests/pecha/test_update.py b/tests/pecha/test_update.py deleted file mode 100644 index 839f6729..00000000 --- a/tests/pecha/test_update.py +++ /dev/null @@ -1,26 +0,0 @@ -from openpecha.pecha import Pecha -from pathlib import Path -from openpecha.utils import read_json, convert_to_base_annotation -import subprocess -from openpecha.pecha.layer import AnnotationType -from openpecha.pecha import get_anns - - - -pecha = Pecha.from_path(Path(f"tests/pecha/update/data/ID8Sv2ynVKZX8wIt")) -annotation_id = "Tm3Uewnh3ySsvgIE" -annotation = [convert_to_base_annotation(ann) for ann in read_json("tests/pecha/update/data/updated_segmentation.json")] -layer_type = AnnotationType.SEGMENTATION - - -def test_update_annotation(): - updated_pecha = pecha.update_annotation(annotation_id=annotation_id, annotation=annotation, layer_type=layer_type) - assert updated_pecha.id == pecha.id - base_name = list(pecha.bases.keys())[0] - ann_store, _ = pecha.get_layer_by_ann_type(base_name=base_name, layer_type=layer_type) - - created_annotations = get_anns(ann_store[0] if isinstance(ann_store, list) else ann_store, include_span=True) - - assert len(created_annotations) == len(annotation) - subprocess.run("rm -rf tests/pecha/update/data/ID8Sv2ynVKZX8wIt", shell=True) - subprocess.run("cp -r tests/pecha/serializers/json/data/ID8Sv2ynVKZX8wIt tests/pecha/update/data/ID8Sv2ynVKZX8wIt", shell=True) \ No newline at end of file diff --git a/tests/pecha/update/data/ID8Sv2ynVKZX8wIt/layers/26E4/segmentation-Tm3Uewnh3ySsvgIE.json b/tests/pecha/update/data/ID8Sv2ynVKZX8wIt/layers/26E4/segmentation-Tm3Uewnh3ySsvgIE.json deleted file mode 100644 index bce267dc..00000000 --- a/tests/pecha/update/data/ID8Sv2ynVKZX8wIt/layers/26E4/segmentation-Tm3Uewnh3ySsvgIE.json +++ /dev/null @@ -1,1320 +0,0 @@ -{ - "@type": "AnnotationStore", - "@id": "ID8Sv2ynVKZX8wIt", - "resources": [ - { - "@type": "TextResource", - "@id": "26E4", - "@include": "../../base/26E4.txt" - } - ], - "annotationsets": [ - { - "@type": "AnnotationDataSet", - "@id": "segmentation_annotation", - "keys": [ - { - "@type": "DataKey", - "@id": "index" - }, - { - "@type": "DataKey", - "@id": "segmentation_type" - } - ], - "data": [ - { - "@type": "AnnotationData", - "@id": "84A8849AB4", - "key": "index", - "value": { - "@type": "Int", - "value": 1 - } - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "key": "segmentation_type", - "value": { - "@type": "String", - "value": "segmentation" - } - }, - { - "@type": "AnnotationData", - "@id": "222D05AA0A", - "key": "index", - "value": { - "@type": "Int", - "value": 2 - } - }, - { - "@type": "AnnotationData", - "@id": "E3785540BA", - "key": "index", - "value": { - "@type": "Int", - "value": 3 - } - }, - { - "@type": "AnnotationData", - "@id": "418F632528", - "key": "index", - "value": { - "@type": "Int", - "value": 4 - } - }, - { - "@type": "AnnotationData", - "@id": "9E813CA691", - "key": "index", - "value": { - "@type": "Int", - "value": 5 - } - }, - { - "@type": "AnnotationData", - "@id": "95ABD8599D", - "key": "index", - "value": { - "@type": "Int", - "value": 6 - } - }, - { - "@type": "AnnotationData", - "@id": "9AE80BBF3E", - "key": "index", - "value": { - "@type": "Int", - "value": 7 - } - }, - { - "@type": "AnnotationData", - "@id": "EC333836BE", - "key": "index", - "value": { - "@type": "Int", - "value": 8 - } - }, - { - "@type": "AnnotationData", - "@id": "A561344039", - "key": "index", - "value": { - "@type": "Int", - "value": 9 - } - }, - { - "@type": "AnnotationData", - "@id": "DBECC4EF6F", - "key": "index", - "value": { - "@type": "Int", - "value": 10 - } - }, - { - "@type": "AnnotationData", - "@id": "4F48498554", - "key": "index", - "value": { - "@type": "Int", - "value": 11 - } - }, - { - "@type": "AnnotationData", - "@id": "4B3DB7A8E7", - "key": "index", - "value": { - "@type": "Int", - "value": 12 - } - }, - { - "@type": "AnnotationData", - "@id": "3DDFE49DED", - "key": "index", - "value": { - "@type": "Int", - "value": 13 - } - }, - { - "@type": "AnnotationData", - "@id": "B6C64F6BBB", - "key": "index", - "value": { - "@type": "Int", - "value": 14 - } - }, - { - "@type": "AnnotationData", - "@id": "FFAF411C2C", - "key": "index", - "value": { - "@type": "Int", - "value": 15 - } - }, - { - "@type": "AnnotationData", - "@id": "232E82AD4C", - "key": "index", - "value": { - "@type": "Int", - "value": 16 - } - }, - { - "@type": "AnnotationData", - "@id": "F290DD99A9", - "key": "index", - "value": { - "@type": "Int", - "value": 17 - } - }, - { - "@type": "AnnotationData", - "@id": "CE5C4051BA", - "key": "index", - "value": { - "@type": "Int", - "value": 18 - } - }, - { - "@type": "AnnotationData", - "@id": "DE143AE56E", - "key": "index", - "value": { - "@type": "Int", - "value": 19 - } - }, - { - "@type": "AnnotationData", - "@id": "D49C52979C", - "key": "index", - "value": { - "@type": "Int", - "value": 20 - } - }, - { - "@type": "AnnotationData", - "@id": "4DF7691F34", - "key": "index", - "value": { - "@type": "Int", - "value": 21 - } - }, - { - "@type": "AnnotationData", - "@id": "0D98A8BED4", - "key": "index", - "value": { - "@type": "Int", - "value": 22 - } - }, - { - "@type": "AnnotationData", - "@id": "BC498F81BD", - "key": "index", - "value": { - "@type": "Int", - "value": 23 - } - }, - { - "@type": "AnnotationData", - "@id": "CC3CCCF793", - "key": "index", - "value": { - "@type": "Int", - "value": 24 - } - }, - { - "@type": "AnnotationData", - "@id": "6661E11FF6", - "key": "index", - "value": { - "@type": "Int", - "value": 25 - } - }, - { - "@type": "AnnotationData", - "@id": "E38E1956F9", - "key": "index", - "value": { - "@type": "Int", - "value": 26 - } - }, - { - "@type": "AnnotationData", - "@id": "91D6998190", - "key": "index", - "value": { - "@type": "Int", - "value": 27 - } - }, - { - "@type": "AnnotationData", - "@id": "34B618B6DD", - "key": "index", - "value": { - "@type": "Int", - "value": 28 - } - }, - { - "@type": "AnnotationData", - "@id": "10F713FAC8", - "key": "index", - "value": { - "@type": "Int", - "value": 29 - } - }, - { - "@type": "AnnotationData", - "@id": "DE82D2441F", - "key": "index", - "value": { - "@type": "Int", - "value": 30 - } - }, - { - "@type": "AnnotationData", - "@id": "C13C1ACA0C", - "key": "index", - "value": { - "@type": "Int", - "value": 31 - } - }, - { - "@type": "AnnotationData", - "@id": "07FB3155B3", - "key": "index", - "value": { - "@type": "Int", - "value": 32 - } - } - ] - } - ], - "annotations": [ - { - "@type": "Annotation", - "@id": "B07549701B", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 0 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 54 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "84A8849AB4", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "60010AB6CD", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 55 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 110 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "222D05AA0A", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "C8B26CFFA4", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 111 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 175 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "E3785540BA", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "5C519C091F", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 176 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 193 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "418F632528", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "A99CFE3DC3", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 194 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 251 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "9E813CA691", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "BCBF7B0B44", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 252 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 287 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "95ABD8599D", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "74CAE409E9", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 288 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 428 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "9AE80BBF3E", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "2E2FB82547", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 429 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 527 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "EC333836BE", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "43ECDEE9B0", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 528 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 669 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "A561344039", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "AA18B6218B", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 670 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 730 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "DBECC4EF6F", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "418BE00783", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 731 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 861 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "4F48498554", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "DA91DA4BE4", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 862 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 1321 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "4B3DB7A8E7", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "B4245065D7", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 1322 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 1362 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "3DDFE49DED", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "0F1030EE43", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 1363 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 1363 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "B6C64F6BBB", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "97F7C93C66", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 1364 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 1435 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "FFAF411C2C", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "3A307BA7F7", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 1436 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 1516 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "232E82AD4C", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "B8EE2A4C35", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 1517 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 1667 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "F290DD99A9", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "80D90F8088", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 1668 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 1888 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "CE5C4051BA", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "7BBBEC254B", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 1889 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 1976 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "DE143AE56E", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "7CE016E497", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 1977 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 2155 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "D49C52979C", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "CEDCB38435", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 2156 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 2269 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "4DF7691F34", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "8FF1A6F3F3", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 2270 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 2366 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "0D98A8BED4", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "671711C44C", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 2367 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 2527 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "BC498F81BD", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "0CFE1EF97C", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 2528 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 2763 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "CC3CCCF793", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "0BFEA4BBEF", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 2764 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 2821 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "6661E11FF6", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "6D8BDFC8FD", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 2822 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 2923 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "E38E1956F9", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "74E5403E03", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 2924 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 3106 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "91D6998190", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "3F2B818452", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 3107 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 3216 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "34B618B6DD", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "D050E18B90", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 3217 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 3262 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "10F713FAC8", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "4544050EE3", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 3263 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 3305 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "DE82D2441F", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "9853B76832", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 3306 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 3548 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "C13C1ACA0C", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - }, - { - "@type": "Annotation", - "@id": "769E1974BB", - "target": { - "@type": "TextSelector", - "resource": "26E4", - "offset": { - "@type": "Offset", - "begin": { - "@type": "BeginAlignedCursor", - "value": 3549 - }, - "end": { - "@type": "BeginAlignedCursor", - "value": 3606 - } - } - }, - "data": [ - { - "@type": "AnnotationData", - "@id": "07FB3155B3", - "set": "segmentation_annotation" - }, - { - "@type": "AnnotationData", - "@id": "B144C11491", - "set": "segmentation_annotation" - } - ] - } - ] -} \ No newline at end of file diff --git a/tests/test_ids.py b/tests/test_ids.py index 037eee7e..69140a7e 100644 --- a/tests/test_ids.py +++ b/tests/test_ids.py @@ -3,8 +3,6 @@ from openpecha.ids import ( get_annotation_id, get_base_id, - get_id, - get_initial_pecha_id, get_layer_id, get_uuid, ) @@ -16,16 +14,6 @@ def test_get_uuid(): r"^[0-9a-fA-F]{32}$", uuid ), f"UUID {uuid} is not in the correct format" - -def test_get_id(): - prefix = "T" - length = 4 - generated_id = get_id(prefix, length) - assert re.match( - r"^T[0-9A-F]{4}$", generated_id - ), f"ID {generated_id} is not in the correct format" - - def test_get_base_id(): base_id = get_base_id() assert re.match( @@ -40,13 +28,6 @@ def test_get_layer_id(): ), f"Layer ID {layer_id} is not in the correct format" -def test_get_initial_pecha_id(): - initial_pecha_id = get_initial_pecha_id() - assert re.match( - r"^I[0-9A-F]{8}$", initial_pecha_id - ), f"Initial Pecha ID {initial_pecha_id} is not in the correct format" - - def test_get_annotation_id(): ann_id = get_annotation_id() assert len(ann_id) == 10 From 556efcf9770e80dd2d81e39c0d7831f02bd73de3 Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 09:53:17 +0530 Subject: [PATCH 02/15] removed index if annotation type is alignment or segmentation --- src/openpecha/pecha/__init__.py | 9 ++++++++- tests/pecha/test_create_pecha.py | 8 +++++--- 2 files changed, 13 insertions(+), 4 deletions(-) diff --git a/src/openpecha/pecha/__init__.py b/src/openpecha/pecha/__init__.py index 834ccadc..04e987a8 100644 --- a/src/openpecha/pecha/__init__.py +++ b/src/openpecha/pecha/__init__.py @@ -77,7 +77,7 @@ def create_pecha(cls, pecha_id: str, base_text: str, annotation_id: str, annotat return pecha - def add(self, annotation_id: str, annotation: List[BaseAnnotation]) -> "Pecha": + def add(self, annotation_id: str, annotation: List[BaseAnnotation], annotation_type: str) -> "Pecha": base_name = next(iter(self.bases)) ann_type = get_annotation_type(annotation) if check_annotation_exists(self.layer_path/base_name/f"{ann_type.value}-{annotation_id}.json"): @@ -190,6 +190,12 @@ def add_annotation( ann_group_type = layer_type.annotation_group_type ann_data[ann_group_type.value] = layer_type.value + if layer_type in [ + AnnotationType.ALIGNMENT, + AnnotationType.SEGMENTATION, + ]: + ann_data.pop("index", None) + start, end = ( annotation.span.start, annotation.span.end, @@ -219,6 +225,7 @@ def add_annotation( raise StamAddAnnotationError( f"[Error] Failed to add annotation to STAM: {e}" ) + return ann_store def set_metadata(self, pecha_metadata: Dict): diff --git a/tests/pecha/test_create_pecha.py b/tests/pecha/test_create_pecha.py index 6f12eff6..dbc5738f 100644 --- a/tests/pecha/test_create_pecha.py +++ b/tests/pecha/test_create_pecha.py @@ -17,16 +17,18 @@ def test_create_pecha(): assert pecha.bases[base_name] == data["base_text"] ann_store, _ = pecha.get_layer_by_ann_type(base_name=base_name, layer_type=AnnotationType.ALIGNMENT) + # ann_store is a list, we need to use the first AnnotationStore created_annotations = get_anns(ann_store[0] if isinstance(ann_store, list) else ann_store, include_span=True) assert len(created_annotations) == len(data["annotation"]) first_created = created_annotations[0] + first_original = data["annotation"][0] assert first_created["span"]["start"] == first_original["span"]["start"] assert first_created["span"]["end"] == first_original["span"]["end"] - assert first_created["index"] == first_original["index"] + assert first_created.get("index") is None assert first_created["alignment_index"] == first_original["alignment_index"] def test_add(): @@ -36,7 +38,7 @@ def test_add(): base_name = next(iter(pecha.bases)) annotation_id = generate_id() - annotation_id = pecha.add(annotation_id=annotation_id, annotation=annotation) + annotation_id = pecha.add(annotation_id=annotation_id, annotation=annotation, annotation_type="alignment") ann_store, _ = pecha.get_layer_by_ann_type(base_name=base_name, layer_type=AnnotationType.ALIGNMENT) @@ -48,7 +50,7 @@ def test_add(): first_original = data["annotation"][0] assert first_created["span"]["start"] == first_original["span"]["start"] assert first_created["span"]["end"] == first_original["span"]["end"] - assert first_created["index"] == first_original["index"] + assert first_created.get("index") is None assert first_created["alignment_index"] == first_original["alignment_index"] # Clean up - remove the added annotation layer to keep test data clean From d55f43c3b121963f09da37d4ad5afe786c2670d9 Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 10:55:20 +0530 Subject: [PATCH 03/15] test parse --- src/openpecha/pecha/annotations.py | 5 +-- src/openpecha/pecha/parsers/edition.py | 3 +- tests/pecha/parser/edition/test_edition.py | 43 +++++++++++----------- 3 files changed, 26 insertions(+), 25 deletions(-) diff --git a/src/openpecha/pecha/annotations.py b/src/openpecha/pecha/annotations.py index 44f48689..5d295b3b 100644 --- a/src/openpecha/pecha/annotations.py +++ b/src/openpecha/pecha/annotations.py @@ -40,7 +40,6 @@ def end_must_not_be_less_than_start(self) -> "span": class BaseAnnotation(BaseModel): span: span - metadata: Optional[Dict] = None model_config = ConfigDict(extra="allow") @@ -54,11 +53,11 @@ def get_dict(self): class SegmentationAnnotation(BaseAnnotation): - index: int + id: str = Field(..., description="Annotation ID") class AlignmentAnnotation(BaseAnnotation): - index: int + id: str = Field(..., description="Annotation ID") alignment_index: list[int] = Field( description="Index of the alignment, which can be of translation or commentary" ) diff --git a/src/openpecha/pecha/parsers/edition.py b/src/openpecha/pecha/parsers/edition.py index 27a028a1..7bb996e5 100644 --- a/src/openpecha/pecha/parsers/edition.py +++ b/src/openpecha/pecha/parsers/edition.py @@ -16,6 +16,7 @@ from openpecha.pecha.layer import AnnotationType from openpecha.pecha.parsers import update_coords from openpecha.pecha.serializers.json import JsonSerializer +from openpecha.ids import get_annotation_id logger = get_logger(__name__) @@ -44,8 +45,8 @@ def parse_segmentation(self, segments: list[str]) -> list[SegmentationAnnotation for index, segment in enumerate(segments, start=1): anns.append( SegmentationAnnotation( + id=str(index), span=span(start=char_count, end=char_count + len(segment)), - index=index, ) ) char_count += len(segment) + 1 diff --git a/tests/pecha/parser/edition/test_edition.py b/tests/pecha/parser/edition/test_edition.py index 847152f3..a0e53fee 100644 --- a/tests/pecha/parser/edition/test_edition.py +++ b/tests/pecha/parser/edition/test_edition.py @@ -36,16 +36,16 @@ def test_segmentation_parse(self): anns = parser.parse_segmentation(segments) expected_anns = [ - SegmentationAnnotation(span=span(start=0, end=87), index=1), - SegmentationAnnotation(span=span(start=88, end=207), index=2), - SegmentationAnnotation(span=span(start=208, end=283), index=3), - SegmentationAnnotation(span=span(start=284, end=361), index=4), - SegmentationAnnotation(span=span(start=362, end=508), index=5), - SegmentationAnnotation(span=span(start=509, end=844), index=6), - SegmentationAnnotation(span=span(start=845, end=1129), index=7), - SegmentationAnnotation(span=span(start=1130, end=1217), index=8), - SegmentationAnnotation(span=span(start=1218, end=1409), index=9), - SegmentationAnnotation(span=span(start=1410, end=1605), index=10), + SegmentationAnnotation(span=span(start=0, end=87), id=None), + SegmentationAnnotation(span=span(start=88, end=207), id=None), + SegmentationAnnotation(span=span(start=208, end=283), id=None), + SegmentationAnnotation(span=span(start=284, end=361), id=None), + SegmentationAnnotation(span=span(start=362, end=508), id=None), + SegmentationAnnotation(span=span(start=509, end=844), id=None), + SegmentationAnnotation(span=span(start=845, end=1129), id=None), + SegmentationAnnotation(span=span(start=1130, end=1217), id=None), + SegmentationAnnotation(span=span(start=1218, end=1409), id=None), + SegmentationAnnotation(span=span(start=1410, end=1605), id=None), ] assert anns == expected_anns @@ -244,70 +244,71 @@ def test_parse(self): ann_store=AnnotationStore(file=str(self.pecha.layer_path / seg_layer_path)), include_span=True, ) + print("SEG_ANNS", seg_anns) expected_seg_anns = [ { - "index": 1, + "id": "1", "segmentation_type": "segmentation", "text": "བུ་མ་འཇུག་པ་ལས་སེམས་བསྐྱེད་དྲུག་པ། ཤོ་ལོ་ཀ ༡-༦༤ མངོན་དུ་ཕྱོགས་པར་མཉམ་བཞག་སེམས་གནས་ཏེ། །", "span": {"start": 0, "end": 87}, }, { - "index": 2, + "id": "2", "segmentation_type": "segmentation", "text": "ྫོགས་པའི་སངས་རྒྱས་ཆོས་ལ་མངོན་ཕྱོགས་ཤིང༌། །འདི་བརྟེན་འབྱུང་བའི་དེ་ཉིད་མཐོང་བ་དེས། །ཤེས་རབ་གནས་པས་འགོག་པ་ཐོབ་པར་འགྱུར། །", "span": {"start": 88, "end": 206}, }, { - "index": 3, + "index": "3", "segmentation_type": "segmentation", "text": "ཇི་ལྟར་ལོང་བའི་ཚོགས་ཀུན་བདེ་བླག་ཏུ། །མིག་ལྡན་སྐྱེས་བུ་གཅིག་གིས་འདོད་པ་ཡི། །", "span": {"start": 207, "end": 282}, }, { - "index": 4, + "id": "4", "segmentation_type": "segmentation", "text": "ུལ་དུ་འཁྲིད་པ་དེ་བཞིན་འདིར་ཡང་བློས། །མིག་ཉམས་ཡོན་ཏན་བླངས་ཏེ་རྒྱལ་ཉིད་འགྲོ། །", "span": {"start": 283, "end": 359}, }, { - "index": 5, + "id": "5", "segmentation_type": "segmentation", "text": "ཇི་ལྟར་དེ་ཡིས་ཆེས་ཟབ་ཆོས་རྟོགས་པ། །ལུང་དང་གཞན་ཡང་རིགས་པས་ཡིན་པས་ན། །དེ་ལྟར་འཕགས་པ་ཀླུ་སྒྲུབ་གཞུང་ལུགས་ལས། །ཇི་ལྟར་གནས་པའི་ལུགས་བཞིན་བརྗོད་པར་བྱ། ", "span": {"start": 360, "end": 505}, }, { - "index": 6, + "id": "6", "segmentation_type": "segmentation", "text": "\nསོ་སོ་སྐྱེ་བོའི་དུས་ནའང་སྟོང་པ་ཉིད་ཐོས་ནས། །ནང་དུ་རབ་ཏུ་དགའ་བ་ཡང་དང་ཡང་དུ་འབྱུང༌། །རབ་ཏུ་དགའ་བ་ལས་བྱུང་མཆི་མས་མིག་བརླན་ཞིང༌། །ལུས་ཀྱི་བ་སྤུ་ལྡང་པར་འགྱུར་པ་གང་ཡིན་པ། །\nདེ་ལ་རྫོགས་པའི་སངས་རྒྱས་བློ་ཡི་ས་བོན་ཡོད། །དེ་ཉིད་ཉེ་བར་བསྟན་པའི་སྣོད་ནི་དེ་ཡིན་ཏེ། །དེ་ལ་དམ་པའི་དོན་གྱི་བདེན་པ་བསྟན་པར་བྱ། །དེ་ལ་དེ་ཡི་རྗེས་སུ་འགྲོ་བའི་ཡོན་ཏན་འབྱུང༌། །\nརྟག་ཏུ་ཚུལ་ཁྲིམས་ཡང་དག་བླངས་ནས་གནས་པར་འག", "span": {"start": 506, "end": 884}, }, { - "index": 7, + "id": "7", "segmentation_type": "segmentation", "text": "ུར། །སྦྱིན་པ་གཏོང་བར་འགྱུར་ཞིང་སྙིང་རྗེ་བསྟེན་པར་བྱེད། །བཟོད་པ་སྒོམ་བྱེད་དེ་ཡི་དགེ་བའང་བྱང་ཆུབ་ཏུ། །འགྲོ་བ་དགྲོལ་བར་བྱ་ཕྱིར་ཡོངས་སུ་བསྔོ་བྱེད་ཅིང༌། །\nརྫོགས་པའི་བྱང་ཆུབ་སེམས་དཔའ་རྣམས་ལ་གུས་པར་བྱེད། །ཟབ་ཅིང་རྒྱ་ཆེའི་ཚུལ་ལ་མཁས་པའི་སྐྱེ་བོས་ནི། །རིམ་གྱིས་རབ་ཏུ་དགའ་བའི་ས་ནི་འཐོབ་འགྱུར་བས།", "span": {"start": 885, "end": 1169}, }, { - "index": 8, + "id": "8", "segmentation_type": "segmentation", "text": "།དེ་ནི་དོན་དུ་གཉེར་བས་ལམ་འདི་མཉན་པར་གྱིས། །", "span": {"start": 1170, "end": 1213}, }, { - "index": 9, + "id": "9", "segmentation_type": "segmentation", "text": "དེ་ཉིད་དེ་ལས་འབྱུང་མིན་གཞན་དག་ལས་ལྟ་ག་ལ་ཞིག །གཉིས་ཀ་ལས་ཀྱང་མ་ཡིན་རྒྱུ་མེད་པར་ནི་ག་ལ་ཡོད། །དེ་ནི་དེ་ལས་འབྱུང་ན་ཡོན་ཏན་འགའ་ཡང་ཡོད་མ་ཡིན། །སྐྱེས་པར་གྱུར་པ་སླར་ཡང་སྐྱེ་བར་རིགས་པའང་མ་ཡིན་ཉིད། །\nསྐྱེ", "span": {"start": 1214, "end": 1407}, }, { - "index": 10, + "id": "10", "segmentation_type": "segmentation", "text": "་ཟིན་སླར་ཡང་སྐྱེ་བར་ཡོངས་སུ་རྟོག་པར་འགྱུར་ན་ནི། །མྱུ་གུ་ལ་སོགས་རྣམས་ཀྱི་སྐྱེ་བ་འདིར་རྙེད་མི་འགྱུར་ཞིང༌། །ས་བོན་སྲིད་མཐར་ཐུག་པར་རབ་ཏུ་སྐྱེ་བ་ཉིད་དུ་འགྱུར། །ཇི་ལྟར་དེ་ཉིད་ཀྱིས་དེ་", "span": {"start": 1408, "end": 1585}, }, ] assert seg_anns == expected_seg_anns - + version_anns = get_anns( ann_store=AnnotationStore(file=str(self.pecha.layer_path / version_path)), include_span=True, From 97c5ae0737b5304a52e0f464099e788fb3034ded Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 11:08:10 +0530 Subject: [PATCH 04/15] fix expected test parse annotation --- tests/pecha/parser/edition/test_edition.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/pecha/parser/edition/test_edition.py b/tests/pecha/parser/edition/test_edition.py index a0e53fee..eccd9312 100644 --- a/tests/pecha/parser/edition/test_edition.py +++ b/tests/pecha/parser/edition/test_edition.py @@ -259,7 +259,7 @@ def test_parse(self): "span": {"start": 88, "end": 206}, }, { - "index": "3", + "id": "3", "segmentation_type": "segmentation", "text": "ཇི་ལྟར་ལོང་བའི་ཚོགས་ཀུན་བདེ་བླག་ཏུ། །མིག་ལྡན་སྐྱེས་བུ་གཅིག་གིས་འདོད་པ་ཡི། །", "span": {"start": 207, "end": 282}, From d3457b12590d20c9c872613b05ae40ebb5b6664b Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 11:13:42 +0530 Subject: [PATCH 05/15] fix test_segmentation in edition parse --- tests/pecha/parser/edition/test_edition.py | 40 +++++++++++----------- 1 file changed, 20 insertions(+), 20 deletions(-) diff --git a/tests/pecha/parser/edition/test_edition.py b/tests/pecha/parser/edition/test_edition.py index eccd9312..86b2697a 100644 --- a/tests/pecha/parser/edition/test_edition.py +++ b/tests/pecha/parser/edition/test_edition.py @@ -36,16 +36,16 @@ def test_segmentation_parse(self): anns = parser.parse_segmentation(segments) expected_anns = [ - SegmentationAnnotation(span=span(start=0, end=87), id=None), - SegmentationAnnotation(span=span(start=88, end=207), id=None), - SegmentationAnnotation(span=span(start=208, end=283), id=None), - SegmentationAnnotation(span=span(start=284, end=361), id=None), - SegmentationAnnotation(span=span(start=362, end=508), id=None), - SegmentationAnnotation(span=span(start=509, end=844), id=None), - SegmentationAnnotation(span=span(start=845, end=1129), id=None), - SegmentationAnnotation(span=span(start=1130, end=1217), id=None), - SegmentationAnnotation(span=span(start=1218, end=1409), id=None), - SegmentationAnnotation(span=span(start=1410, end=1605), id=None), + SegmentationAnnotation(span=span(start=0, end=87), id="1"), + SegmentationAnnotation(span=span(start=88, end=207), id="2"), + SegmentationAnnotation(span=span(start=208, end=283), id="3"), + SegmentationAnnotation(span=span(start=284, end=361), id="4"), + SegmentationAnnotation(span=span(start=362, end=508), id="5"), + SegmentationAnnotation(span=span(start=509, end=844), id="6"), + SegmentationAnnotation(span=span(start=845, end=1129), id="7"), + SegmentationAnnotation(span=span(start=1130, end=1217), id="8"), + SegmentationAnnotation(span=span(start=1218, end=1409), id="9"), + SegmentationAnnotation(span=span(start=1410, end=1605), id="10"), ] assert anns == expected_anns @@ -59,34 +59,34 @@ def test_segmentation_parse(self): updated_anns = update_coords(anns, old_base, new_base) expected_updated_anns = [ SegmentationAnnotation( - span=span(start=0, end=87, errors=None), metadata=None, index=1 + span=span(start=0, end=87, errors=None), id="1" ), SegmentationAnnotation( - span=span(start=88, end=208, errors=None), metadata=None, index=2 + span=span(start=88, end=208, errors=None), id="2" ), SegmentationAnnotation( - span=span(start=209, end=284, errors=None), metadata=None, index=3 + span=span(start=209, end=284, errors=None), id="3" ), SegmentationAnnotation( - span=span(start=285, end=363, errors=None), metadata=None, index=4 + span=span(start=285, end=363, errors=None), id="4" ), SegmentationAnnotation( - span=span(start=364, end=511, errors=None), metadata=None, index=5 + span=span(start=364, end=511, errors=None), id="5" ), SegmentationAnnotation( - span=span(start=512, end=843, errors=None), metadata=None, index=6 + span=span(start=512, end=843, errors=None), id="6" ), SegmentationAnnotation( - span=span(start=843, end=1089, errors=None), metadata=None, index=7 + span=span(start=843, end=1089, errors=None), id="7" ), SegmentationAnnotation( - span=span(start=1090, end=1221, errors=None), metadata=None, index=8 + span=span(start=1090, end=1221, errors=None), id="8" ), SegmentationAnnotation( - span=span(start=1222, end=1411, errors=None), metadata=None, index=9 + span=span(start=1222, end=1411, errors=None), id="9" ), SegmentationAnnotation( - span=span(start=1412, end=1626, errors=None), metadata=None, index=10 + span=span(start=1412, end=1626, errors=None), id="10" ), ] From bc647787a054750aa440a4373da45a092580924a Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 15:06:01 +0530 Subject: [PATCH 06/15] change operation=Enum selection rather than magic string --- tests/pecha/parser/edition/test_edition.py | 45 +++++++++++----------- 1 file changed, 23 insertions(+), 22 deletions(-) diff --git a/tests/pecha/parser/edition/test_edition.py b/tests/pecha/parser/edition/test_edition.py index 86b2697a..62451143 100644 --- a/tests/pecha/parser/edition/test_edition.py +++ b/tests/pecha/parser/edition/test_edition.py @@ -14,6 +14,7 @@ from openpecha.pecha.parsers.edition import EditionParser from openpecha.pecha.serializers.json import JsonSerializer from openpecha.utils import read_json +from openpecha.pecha.annotations import VersionVariantOperations class TestEditionParser(TestCase): @@ -100,40 +101,40 @@ def test_version_parse(self): new_base = "Hello World" diffs = parser.parse_version(old_base, new_base) assert diffs == [ - Version(span=span(start=5, end=5), operation="insertion", text=" World") + Version(span=span(start=5, end=5), operation=VersionVariantOperations.INSERTION, text=" World") ] # Deletion old_base = "Hello World" new_base = "Hello" diffs = parser.parse_version(old_base, new_base) - assert diffs == [Version(span=span(start=5, end=11), operation="deletion")] + assert diffs == [Version(span=span(start=5, end=11), operation=VersionVariantOperations.DELETION)] # Insertion in Between old_base = "Hello World" new_base = "Hello!! World" diffs = parser.parse_version(old_base, new_base) assert diffs == [ - Version(span=span(start=5, end=5), operation="insertion", text="!!") + Version(span=span(start=5, end=5), operation=VersionVariantOperations.INSERTION, text="!!") ] # Deletion in Between old_base = "Good morning, Everyone" new_base = "Good Everyone" diffs = parser.parse_version(old_base, new_base) - assert diffs == [Version(span=span(start=4, end=13), operation="deletion")] + assert diffs == [Version(span=span(start=4, end=13), operation=VersionVariantOperations.DELETION)] # Insertion and Deletion old_base = "Good morning, Ladies and Gentlemen" new_base = "Good Attractive Ladies and Gentlemen" diffs = parser.parse_version(old_base, new_base) assert diffs == [ - Version(span=span(start=5, end=13), operation="deletion"), + Version(span=span(start=5, end=13), operation=VersionVariantOperations.DELETION), Version( - span=span(start=13, end=13), operation="insertion", text="Attractive" + span=span(start=13, end=13), operation=VersionVariantOperations.INSERTION, text="Attractive" ), ] - + # Test with google docs old_basename = list(self.pecha.bases.keys())[0] old_base = self.pecha.get_base(old_basename) @@ -145,91 +146,91 @@ def test_version_parse(self): Version( span=span(start=87, end=87, errors=None), metadata=None, - operation="insertion", + operation=VersionVariantOperations.INSERTION, text="\n", ), Version( span=span(start=282, end=282, errors=None), metadata=None, - operation="insertion", + operation=VersionVariantOperations.INSERTION, text="\n", ), Version( span=span(start=673, end=674, errors=None), metadata=None, - operation="deletion", + operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=888, end=888, errors=None), metadata=None, - operation="insertion", + operation=VersionVariantOperations.INSERTION, text=" རྟག་ཏུ་ཚུལ་ཁྲིམས་ཡང་དག་བླངས་ནས་གནས་པར་འགྱུར།", ), Version( span=span(start=1034, end=1080, errors=None), metadata=None, - operation="deletion", + operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=1080, end=1080, errors=None), metadata=None, - operation="insertion", + operation=VersionVariantOperations.INSERTION, text="འགྲོ་བ་དགྲོལ་བར་བྱ་ཕྱིར་ཡོངས་སུ་བསྔོ་བྱེད་ཅིང༌", ), Version( span=span(start=1083, end=1083, errors=None), metadata=None, - operation="insertion", + operation=VersionVariantOperations.INSERTION, text="\n", ), Version( span=span(start=1170, end=1213, errors=None), metadata=None, - operation="deletion", + operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=1279, end=1279, errors=None), metadata=None, - operation="insertion", + operation=VersionVariantOperations.INSERTION, text="པར་", ), Version( span=span(start=1322, end=1323, errors=None), metadata=None, - operation="deletion", + operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=1323, end=1323, errors=None), metadata=None, - operation="insertion", + operation=VersionVariantOperations.INSERTION, text="བ", ), Version( span=span(start=1441, end=1444, errors=None), metadata=None, - operation="deletion", + operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=1497, end=1500, errors=None), metadata=None, - operation="deletion", + operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=1573, end=1585, errors=None), metadata=None, - operation="deletion", + operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=1616, end=1617, errors=None), metadata=None, - operation="deletion", + operation=VersionVariantOperations.DELETION, text="", ), ] From 5d0c6e850e69688711147faeb906411561922193 Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 15:09:43 +0530 Subject: [PATCH 07/15] fixed test_version test case --- tests/pecha/parser/edition/test_edition.py | 20 +++----------------- 1 file changed, 3 insertions(+), 17 deletions(-) diff --git a/tests/pecha/parser/edition/test_edition.py b/tests/pecha/parser/edition/test_edition.py index 62451143..b3386939 100644 --- a/tests/pecha/parser/edition/test_edition.py +++ b/tests/pecha/parser/edition/test_edition.py @@ -134,7 +134,7 @@ def test_version_parse(self): span=span(start=13, end=13), operation=VersionVariantOperations.INSERTION, text="Attractive" ), ] - + # Test with google docs old_basename = list(self.pecha.bases.keys())[0] old_base = self.pecha.get_base(old_basename) @@ -142,94 +142,80 @@ def test_version_parse(self): segments = self.txt_file.read_text(encoding="utf-8").splitlines() new_base = "\n".join(segments) diffs = parser.parse_version(old_base, new_base) + assert diffs == [ Version( span=span(start=87, end=87, errors=None), - metadata=None, operation=VersionVariantOperations.INSERTION, text="\n", ), Version( span=span(start=282, end=282, errors=None), - metadata=None, operation=VersionVariantOperations.INSERTION, text="\n", ), Version( span=span(start=673, end=674, errors=None), - metadata=None, operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=888, end=888, errors=None), - metadata=None, operation=VersionVariantOperations.INSERTION, text=" རྟག་ཏུ་ཚུལ་ཁྲིམས་ཡང་དག་བླངས་ནས་གནས་པར་འགྱུར།", ), Version( span=span(start=1034, end=1080, errors=None), - metadata=None, operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=1080, end=1080, errors=None), - metadata=None, operation=VersionVariantOperations.INSERTION, text="འགྲོ་བ་དགྲོལ་བར་བྱ་ཕྱིར་ཡོངས་སུ་བསྔོ་བྱེད་ཅིང༌", ), Version( span=span(start=1083, end=1083, errors=None), - metadata=None, operation=VersionVariantOperations.INSERTION, text="\n", ), Version( span=span(start=1170, end=1213, errors=None), - metadata=None, operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=1279, end=1279, errors=None), - metadata=None, operation=VersionVariantOperations.INSERTION, text="པར་", ), Version( span=span(start=1322, end=1323, errors=None), - metadata=None, operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=1323, end=1323, errors=None), - metadata=None, operation=VersionVariantOperations.INSERTION, text="བ", ), Version( span=span(start=1441, end=1444, errors=None), - metadata=None, operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=1497, end=1500, errors=None), - metadata=None, operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=1573, end=1585, errors=None), - metadata=None, operation=VersionVariantOperations.DELETION, text="", ), Version( span=span(start=1616, end=1617, errors=None), - metadata=None, operation=VersionVariantOperations.DELETION, text="", ), @@ -245,7 +231,7 @@ def test_parse(self): ann_store=AnnotationStore(file=str(self.pecha.layer_path / seg_layer_path)), include_span=True, ) - print("SEG_ANNS", seg_anns) + expected_seg_anns = [ { "id": "1", From 1b9dfdae31f3c6f7249e85cc385389daf184daa9 Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 15:57:48 +0530 Subject: [PATCH 08/15] added id and span information in return part when creating pecha --- src/openpecha/pecha/__init__.py | 11 ++++++++++- tests/pecha/test_create_pecha.py | 2 +- 2 files changed, 11 insertions(+), 2 deletions(-) diff --git a/src/openpecha/pecha/__init__.py b/src/openpecha/pecha/__init__.py index 04e987a8..84f4be05 100644 --- a/src/openpecha/pecha/__init__.py +++ b/src/openpecha/pecha/__init__.py @@ -27,7 +27,7 @@ def __init__(self, pecha_id: str, pecha_path: Path) -> None: self.pecha_path = pecha_path self.metadata = self.load_metadata() self.bases = self.load_bases() - # self.annotations = self.load_annotations() + self.annotations = [] @classmethod def from_path(cls, pecha_path: Path) -> "Pecha": @@ -74,6 +74,15 @@ def create_pecha(cls, pecha_id: str, base_text: str, annotation_id: str, annotat for single_annotation in annotation: ann_store = pecha.add_annotation(ann_store=ann_store, annotation=single_annotation, layer_type=ann_type) ann_store.save() + annotations = get_anns(ann_store, include_span=True) + for annotation in annotations: + pecha.annotations.append({ + "span": { + "start": annotation["span"]["start"], + "end": annotation["span"]["end"], + }, + "id": annotation["id"] + }) return pecha diff --git a/tests/pecha/test_create_pecha.py b/tests/pecha/test_create_pecha.py index dbc5738f..79b79254 100644 --- a/tests/pecha/test_create_pecha.py +++ b/tests/pecha/test_create_pecha.py @@ -10,7 +10,7 @@ def test_create_pecha(): annotation = [convert_to_base_annotation(ann) for ann in data["annotation"]] annotation_id = generate_id() pecha = Pecha.create_pecha(pecha_id=data["pecha_id"], base_text=data["base_text"], annotation_id=annotation_id, annotation=annotation) - + # assert pecha.id == data["pecha_id"] base_name = list(pecha.bases.keys())[0] From 6a7b33380bb7712b043dce9eff6811ffa874e8dd Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 16:01:22 +0530 Subject: [PATCH 09/15] check index not present in annotation --- tests/pecha/test_create_pecha.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/tests/pecha/test_create_pecha.py b/tests/pecha/test_create_pecha.py index 79b79254..5eba1fc0 100644 --- a/tests/pecha/test_create_pecha.py +++ b/tests/pecha/test_create_pecha.py @@ -28,7 +28,7 @@ def test_create_pecha(): first_original = data["annotation"][0] assert first_created["span"]["start"] == first_original["span"]["start"] assert first_created["span"]["end"] == first_original["span"]["end"] - assert first_created.get("index") is None + assert not first_created.get("index") assert first_created["alignment_index"] == first_original["alignment_index"] def test_add(): @@ -47,10 +47,11 @@ def test_add(): assert len(created_annotations) == len(data["annotation"]) first_created = created_annotations[0] + first_original = data["annotation"][0] assert first_created["span"]["start"] == first_original["span"]["start"] assert first_created["span"]["end"] == first_original["span"]["end"] - assert first_created.get("index") is None + assert not first_created.get("index") assert first_created["alignment_index"] == first_original["alignment_index"] # Clean up - remove the added annotation layer to keep test data clean From d3a530ff0b910171abd66913b8c0ebe8e454b166 Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 16:10:57 +0530 Subject: [PATCH 10/15] removed pip install -e '.[github]' causing test issue on github workflow --- .github/workflows/CI.yml | 1 - 1 file changed, 1 deletion(-) diff --git a/.github/workflows/CI.yml b/.github/workflows/CI.yml index cb33fa5c..f39e15dc 100644 --- a/.github/workflows/CI.yml +++ b/.github/workflows/CI.yml @@ -25,7 +25,6 @@ jobs: - name: Install dependencies run: | pip install -U pip - pip install -e ".[github]" pip install ".[dev]" - name: Test with pytest From d2214fc65d1ed29e288f0ad2233a5197024e5211 Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 16:13:25 +0530 Subject: [PATCH 11/15] added back the pip install -e '.[github] ' --- .github/workflows/CI.yml | 1 + 1 file changed, 1 insertion(+) diff --git a/.github/workflows/CI.yml b/.github/workflows/CI.yml index f39e15dc..cb33fa5c 100644 --- a/.github/workflows/CI.yml +++ b/.github/workflows/CI.yml @@ -25,6 +25,7 @@ jobs: - name: Install dependencies run: | pip install -U pip + pip install -e ".[github]" pip install ".[dev]" - name: Test with pytest From 7d30d8a1dee49e0ce808b929543296dcb8b0a036 Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 16:17:40 +0530 Subject: [PATCH 12/15] added print statement to check first_created --- tests/pecha/test_create_pecha.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/pecha/test_create_pecha.py b/tests/pecha/test_create_pecha.py index 5eba1fc0..1fb2badd 100644 --- a/tests/pecha/test_create_pecha.py +++ b/tests/pecha/test_create_pecha.py @@ -47,7 +47,7 @@ def test_add(): assert len(created_annotations) == len(data["annotation"]) first_created = created_annotations[0] - + print("FIRST CREATED: ", first_created) first_original = data["annotation"][0] assert first_created["span"]["start"] == first_original["span"]["start"] assert first_created["span"]["end"] == first_original["span"]["end"] From 370086409648dfeeed4828bef443b852a2d0d164 Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 16:32:03 +0530 Subject: [PATCH 13/15] removed popping of index since already removed from segmentation and alignment class model --- src/openpecha/pecha/__init__.py | 6 ------ tests/pecha/test_create_pecha.py | 4 +++- 2 files changed, 3 insertions(+), 7 deletions(-) diff --git a/src/openpecha/pecha/__init__.py b/src/openpecha/pecha/__init__.py index 84f4be05..1ecef65e 100644 --- a/src/openpecha/pecha/__init__.py +++ b/src/openpecha/pecha/__init__.py @@ -199,12 +199,6 @@ def add_annotation( ann_group_type = layer_type.annotation_group_type ann_data[ann_group_type.value] = layer_type.value - if layer_type in [ - AnnotationType.ALIGNMENT, - AnnotationType.SEGMENTATION, - ]: - ann_data.pop("index", None) - start, end = ( annotation.span.start, annotation.span.end, diff --git a/tests/pecha/test_create_pecha.py b/tests/pecha/test_create_pecha.py index 1fb2badd..6ea0ab4a 100644 --- a/tests/pecha/test_create_pecha.py +++ b/tests/pecha/test_create_pecha.py @@ -18,6 +18,8 @@ def test_create_pecha(): ann_store, _ = pecha.get_layer_by_ann_type(base_name=base_name, layer_type=AnnotationType.ALIGNMENT) + print("ANNS STORE: ", ann_store) + # ann_store is a list, we need to use the first AnnotationStore created_annotations = get_anns(ann_store[0] if isinstance(ann_store, list) else ann_store, include_span=True) @@ -47,7 +49,7 @@ def test_add(): assert len(created_annotations) == len(data["annotation"]) first_created = created_annotations[0] - print("FIRST CREATED: ", first_created) + first_original = data["annotation"][0] assert first_created["span"]["start"] == first_original["span"]["start"] assert first_created["span"]["end"] == first_original["span"]["end"] From 1d4cb66e1de69d17d92d7e4da28ad3b9ab038689 Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 16:44:35 +0530 Subject: [PATCH 14/15] removed index from BaseAnnotation class itself --- src/openpecha/pecha/__init__.py | 1 - src/openpecha/pecha/annotations.py | 1 + 2 files changed, 1 insertion(+), 1 deletion(-) diff --git a/src/openpecha/pecha/__init__.py b/src/openpecha/pecha/__init__.py index 1ecef65e..736b2f09 100644 --- a/src/openpecha/pecha/__init__.py +++ b/src/openpecha/pecha/__init__.py @@ -198,7 +198,6 @@ def add_annotation( # Add Annotation Group Type ann_group_type = layer_type.annotation_group_type ann_data[ann_group_type.value] = layer_type.value - start, end = ( annotation.span.start, annotation.span.end, diff --git a/src/openpecha/pecha/annotations.py b/src/openpecha/pecha/annotations.py index 5d295b3b..f5eb9883 100644 --- a/src/openpecha/pecha/annotations.py +++ b/src/openpecha/pecha/annotations.py @@ -47,6 +47,7 @@ def get_dict(self): res = self.model_dump() # Remove span from the dictionary res.pop("span") + res.pop("index") # Remove None values from the dictionary res = {k: v for k, v in res.items() if v is not None} return res From 14d83dd53341020c03a38313a59c8e8f2e6abc69 Mon Sep 17 00:00:00 2001 From: tentse Date: Wed, 22 Oct 2025 17:07:59 +0530 Subject: [PATCH 15/15] removed annotation layer distinguisher function --- src/openpecha/pecha/__init__.py | 12 +++++------- src/openpecha/pecha/annotations.py | 1 - src/openpecha/utils.py | 2 +- tests/pecha/test_create_pecha.py | 12 +++++------- 4 files changed, 11 insertions(+), 16 deletions(-) diff --git a/src/openpecha/pecha/__init__.py b/src/openpecha/pecha/__init__.py index 736b2f09..f5558472 100644 --- a/src/openpecha/pecha/__init__.py +++ b/src/openpecha/pecha/__init__.py @@ -65,14 +65,12 @@ def create(cls, output_path: Optional[Path] = None, pecha_id: Optional[str] = No return cls(pecha_id, pecha_path) @classmethod - def create_pecha(cls, pecha_id: str, base_text: str, annotation_id: str, annotation: List[BaseAnnotation]) -> "Pecha": + def create_pecha(cls, pecha_id: str, base_text: str, annotation_id: str, annotation: List[BaseAnnotation], annotation_type: AnnotationType) -> "Pecha": pecha = cls.create(pecha_id=pecha_id) base_name = pecha.set_base(base_text) - ann_type = get_annotation_type(annotation) - ann_store, _ = pecha.add_layer(base_name=base_name, layer_type=ann_type, annotation_id=annotation_id) - + ann_store, _ = pecha.add_layer(base_name=base_name, layer_type=annotation_type, annotation_id=annotation_id) for single_annotation in annotation: - ann_store = pecha.add_annotation(ann_store=ann_store, annotation=single_annotation, layer_type=ann_type) + ann_store = pecha.add_annotation(ann_store=ann_store, annotation=single_annotation, layer_type=annotation_type) ann_store.save() annotations = get_anns(ann_store, include_span=True) for annotation in annotations: @@ -86,9 +84,9 @@ def create_pecha(cls, pecha_id: str, base_text: str, annotation_id: str, annotat return pecha - def add(self, annotation_id: str, annotation: List[BaseAnnotation], annotation_type: str) -> "Pecha": + def add(self, annotation_id: str, annotation: List[BaseAnnotation], annotation_type: AnnotationType) -> "Pecha": base_name = next(iter(self.bases)) - ann_type = get_annotation_type(annotation) + ann_type = annotation_type if check_annotation_exists(self.layer_path/base_name/f"{ann_type.value}-{annotation_id}.json"): raise ValueError(f"Annotation with id {annotation_id} already exists") ann_store, _ = self.add_layer(base_name=base_name, layer_type=ann_type, annotation_id=annotation_id) diff --git a/src/openpecha/pecha/annotations.py b/src/openpecha/pecha/annotations.py index f5eb9883..5d295b3b 100644 --- a/src/openpecha/pecha/annotations.py +++ b/src/openpecha/pecha/annotations.py @@ -47,7 +47,6 @@ def get_dict(self): res = self.model_dump() # Remove span from the dictionary res.pop("span") - res.pop("index") # Remove None values from the dictionary res = {k: v for k, v in res.items() if v is not None} return res diff --git a/src/openpecha/utils.py b/src/openpecha/utils.py index 750b8a49..ceb701c6 100644 --- a/src/openpecha/utils.py +++ b/src/openpecha/utils.py @@ -58,5 +58,5 @@ def write_json( def convert_to_base_annotation(raw_annotation): span_data = raw_annotation["span"] annotation_span = span(start=span_data["start"], end=span_data["end"]) - annotation_data = {k: v for k, v in raw_annotation.items() if k != "span"} + annotation_data = {k: v for k, v in raw_annotation.items() if k != "span" and k != "index"} return BaseAnnotation(span=annotation_span, **annotation_data) \ No newline at end of file diff --git a/tests/pecha/test_create_pecha.py b/tests/pecha/test_create_pecha.py index 6ea0ab4a..26f0e4d9 100644 --- a/tests/pecha/test_create_pecha.py +++ b/tests/pecha/test_create_pecha.py @@ -1,4 +1,4 @@ -from openpecha.pecha import Pecha, get_anns, get_annotation_type +from openpecha.pecha import Pecha, get_anns from openpecha.utils import read_json, convert_to_base_annotation from pathlib import Path from openpecha.pecha.layer import AnnotationType @@ -9,7 +9,7 @@ def test_create_pecha(): data = read_json("tests/pecha/data/ITEST001.json") annotation = [convert_to_base_annotation(ann) for ann in data["annotation"]] annotation_id = generate_id() - pecha = Pecha.create_pecha(pecha_id=data["pecha_id"], base_text=data["base_text"], annotation_id=annotation_id, annotation=annotation) + pecha = Pecha.create_pecha(pecha_id=data["pecha_id"], base_text=data["base_text"], annotation_id=annotation_id, annotation=annotation, annotation_type=AnnotationType.ALIGNMENT) # assert pecha.id == data["pecha_id"] @@ -18,8 +18,6 @@ def test_create_pecha(): ann_store, _ = pecha.get_layer_by_ann_type(base_name=base_name, layer_type=AnnotationType.ALIGNMENT) - print("ANNS STORE: ", ann_store) - # ann_store is a list, we need to use the first AnnotationStore created_annotations = get_anns(ann_store[0] if isinstance(ann_store, list) else ann_store, include_span=True) @@ -40,7 +38,7 @@ def test_add(): base_name = next(iter(pecha.bases)) annotation_id = generate_id() - annotation_id = pecha.add(annotation_id=annotation_id, annotation=annotation, annotation_type="alignment") + annotation_id = pecha.add(annotation_id=annotation_id, annotation=annotation, annotation_type=AnnotationType.ALIGNMENT) ann_store, _ = pecha.get_layer_by_ann_type(base_name=base_name, layer_type=AnnotationType.ALIGNMENT) @@ -49,7 +47,7 @@ def test_add(): assert len(created_annotations) == len(data["annotation"]) first_created = created_annotations[0] - + first_original = data["annotation"][0] assert first_created["span"]["start"] == first_original["span"]["start"] assert first_created["span"]["end"] == first_original["span"]["end"] @@ -57,7 +55,7 @@ def test_add(): assert first_created["alignment_index"] == first_original["alignment_index"] # Clean up - remove the added annotation layer to keep test data clean - ann_type = get_annotation_type(annotation) + ann_type = AnnotationType.ALIGNMENT annotation_layer_file = pecha.layer_path / base_name / f"{ann_type.value}-{annotation_id}.json" if annotation_layer_file.exists(): annotation_layer_file.unlink() \ No newline at end of file