Skip to content

Commit

Permalink
WIP: Add functions for pack/unpack of all identifiers at once
Browse files Browse the repository at this point in the history
  • Loading branch information
jacksonj04 committed Nov 25, 2024
1 parent 5f648e5 commit d931400
Show file tree
Hide file tree
Showing 2 changed files with 75 additions and 4 deletions.
33 changes: 29 additions & 4 deletions src/caselawclient/models/documents/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
from ds_caselaw_utils import courts
from ds_caselaw_utils.courts import CourtNotFoundException
from ds_caselaw_utils.types import NeutralCitationString
from lxml import etree
from lxml import html as html_parser
from requests_toolbelt.multipart import decoder

Expand All @@ -15,6 +16,7 @@
NotSupportedOnVersion,
OnlySupportedOnVersion,
)
from caselawclient.models.identifiers.unpacker import unpack_identifier_from_etree
from caselawclient.models.utilities import VersionsDict, extract_version, render_versions
from caselawclient.models.utilities.aws import (
ParserInstructionsDict,
Expand Down Expand Up @@ -125,6 +127,8 @@ class Document:
Individual document classes should extend this list where necessary to validate document type-specific attributes.
"""

identifiers: dict[str, Identifier] = {}

def __init__(self, uri: DocumentURIString, api_client: "MarklogicApiClient", search_query: Optional[str] = None):
"""
:param uri: The URI of the document to retrieve from MarkLogic.
Expand All @@ -147,6 +151,8 @@ def __init__(self, uri: DocumentURIString, api_client: "MarklogicApiClient", sea
)
""" `Document.body` represents the body of the document itself, without any information such as version tracking or properties. """

self.identifiers = {}

def __repr__(self) -> str:
name = self.body.name or "un-named"
return f"<{self.document_noun} {self.uri}: {name}>"
Expand All @@ -161,10 +167,29 @@ def docx_exists(self) -> bool:
"""There is a docx in S3 private bucket for this Document"""
return check_docx_exists(self.uri)

@property
def identifiers(self) -> list[Identifier]:
"""A list of all known identifiers for this document."""
return []
def _load_identifiers(self) -> None:
"""Load this document's identifiers from MarkLogic"""
identifiers_element_as_string = self.api_client.get_property(self.uri, "identifiers")
identifiers_element_as_etree = etree.fromstring(identifiers_element_as_string)

for identifier_etree in identifiers_element_as_etree.findall("identifier"):
identifier = unpack_identifier_from_etree(identifier_etree)
self.add_identifier(identifier)

def add_identifier(self, identifier: Identifier) -> None:
"""Add an identifier to this Document's identifiers array."""
self.identifiers[identifier.uuid] = identifier

def identifiers_as_etree(self) -> etree._Element:
identifiers_root = etree.Element("identifiers")

for identifier in self.identifiers.values():
identifiers_root.append(identifier.as_xml_tree)

return identifiers_root

def save_identifiers(self) -> None:
"""Save the current state of this Document's identifiers to MarkLogic."""

@property
def best_human_identifier(self) -> Optional[str]:
Expand Down
46 changes: 46 additions & 0 deletions tests/models/documents/test_document_identifiers.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
from lxml import etree

from caselawclient.factories import DocumentFactory
from tests.models.identifiers.test_identifiers import TestIdentifier


class TestDocumentIdentifiers:
def test_add_identifiers(self):
document = DocumentFactory.build()

identifier_1 = TestIdentifier(uuid="e28e3ef1-85ed-4997-87ee-e7428a6cc02e", value="TEST-123")
identifier_2 = TestIdentifier(uuid="14ce4b3b-03c8-44f9-a29e-e02ce35fe136", value="TEST-456")
document.add_identifier(identifier_1)
document.add_identifier(identifier_2)

assert document.identifiers == {
"e28e3ef1-85ed-4997-87ee-e7428a6cc02e": identifier_1,
"14ce4b3b-03c8-44f9-a29e-e02ce35fe136": identifier_2,
}

def test_identifiers_as_etree(self):
document = DocumentFactory.build()

identifier_1 = TestIdentifier(uuid="e28e3ef1-85ed-4997-87ee-e7428a6cc02e", value="TEST-123")
identifier_2 = TestIdentifier(uuid="14ce4b3b-03c8-44f9-a29e-e02ce35fe136", value="TEST-456")
document.add_identifier(identifier_1)
document.add_identifier(identifier_2)

expected_xml = """
<identifiers>
<identifier>
<namespace>test</namespace>
<uuid>e28e3ef1-85ed-4997-87ee-e7428a6cc02e</uuid>
<value>TEST-123</value>
</identifier>
<identifier>
<namespace>test</namespace>
<uuid>14ce4b3b-03c8-44f9-a29e-e02ce35fe136</uuid>
<value>TEST-456</value>
</identifier>
</identifiers>
"""

assert etree.canonicalize(document.identifiers_as_etree(), strip_text=True) == etree.canonicalize(
etree.fromstring(expected_xml), strip_text=True
)

0 comments on commit d931400

Please sign in to comment.