Skip to content
Merged
Show file tree
Hide file tree
Changes from 1 commit
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 22 additions & 3 deletions packtools/sps/formats/pdf/pipeline/xml.py
Original file line number Diff line number Diff line change
Expand Up @@ -112,6 +112,7 @@ def extract_contrib_data(xml_tree):
contrib_group = xml_tree.find('.//contrib-group')
if contrib_group is not None:
aff_mapping = {}
referenced_aff_ids = set()

for aff in xml_tree.findall('.//aff'):
aff_id = aff.get('id')
Expand Down Expand Up @@ -140,12 +141,30 @@ def extract_contrib_data(xml_tree):
full_name += label
full_name += corresp_mark
authors_names.append(full_name)

for aff in xml_tree.findall('.//aff'):

for xref in contrib.findall('.//xref[@ref-type="aff"]'):
rid = xref.get('rid')
if rid:
referenced_aff_ids.add(rid)

# Some SciELO packages carry duplicate/orphaned <aff> elements (seen
# in 7/26 of the real test corpus) that aren't referenced by any
# contributor's xref - e.g. a second, unused copy of the same
# affiliations under ids like aff1e/aff0100/aff1s, or (in one case)
# a full second set aff13-aff24 duplicating aff1-aff12. The id
# suffix convention isn't consistent enough to filter on, but none
# of these duplicates are ever cited, so restrict the printed list
# to affiliations actually referenced. Falls back to every <aff> if
# none are referenced at all (defensive; not seen in practice).
affs_to_print = xml_tree.findall('.//aff')
if referenced_aff_ids:
affs_to_print = [aff for aff in affs_to_print if aff.get('id') in referenced_aff_ids]

for aff in affs_to_print:
label = aff.find('label').text if aff.find('label') is not None else ''
institution = aff.find('institution[@content-type="original"]')
institution_name = institution.text if institution is not None else ''

if institution_name:
aff_info = f"{label}[^] {institution_name}"
affiliations.append(aff_info)
Expand Down
93 changes: 93 additions & 0 deletions tests/sps/formats/pdf/pipeline/test_xml.py
Original file line number Diff line number Diff line change
Expand Up @@ -644,6 +644,99 @@ def test_extract_contrib_data_missing_label(self):
self.assertEqual(result['authors_names'], ['John Smith[^]'])
self.assertEqual(result['affiliations'], ['[^] University X'])

def test_unreferenced_duplicate_affiliation_is_not_printed(self):
# Regression: some SciELO packages carry a duplicate/orphaned <aff>
# that no contributor's xref points to (seen in a9.xml of the real
# test corpus: aff1e/aff2e duplicate aff1/aff2 under a different id
# suffix). findall('.//aff') picked up every <aff> in the document
# regardless of whether any contrib actually cited it, so the
# affiliation printed twice.
xml = etree.fromstring("""
<article>
<contrib-group>
<contrib>
<name>
<surname>Smith</surname>
<given-names>John</given-names>
</name>
<xref ref-type="aff" rid="aff1"/>
</contrib>
</contrib-group>
<aff id="aff1">
<label>I</label>
<institution content-type="original">University A</institution>
</aff>
<aff id="aff1e">
<label>I</label>
<institution content-type="original">University A</institution>
</aff>
</article>
""")
result = xml_pipe.extract_contrib_data(xml)
self.assertEqual(result['affiliations'], ['I[^] University A'])

def test_all_referenced_affiliations_are_kept_regardless_of_id_pattern(self):
# The duplicate <aff>'s id doesn't follow one fixed naming
# convention across real packages (seen: aff1e, aff0100, aff1s, and
# a fully separate aff13-aff24 numbering) - the fix must not assume
# one, and must not drop a legitimately different affiliation that
# simply happens to use an unusual id.
xml = etree.fromstring("""
<article>
<contrib-group>
<contrib>
<name>
<surname>Smith</surname>
<given-names>John</given-names>
</name>
<xref ref-type="aff" rid="aff01"/>
</contrib>
<contrib>
<name>
<surname>Doe</surname>
<given-names>Jane</given-names>
</name>
<xref ref-type="aff" rid="aff0100"/>
</contrib>
</contrib-group>
<aff id="aff01">
<label>1</label>
<institution content-type="original">University A</institution>
</aff>
<aff id="aff0100">
<label>2</label>
<institution content-type="original">University B</institution>
</aff>
</article>
""")
result = xml_pipe.extract_contrib_data(xml)
self.assertEqual(
result['affiliations'],
['1[^] University A', '2[^] University B'],
)

def test_falls_back_to_every_affiliation_when_none_are_referenced(self):
# Defensive: an article with no aff xrefs at all (not seen in the
# real corpus) must not end up with an empty affiliation list.
xml = etree.fromstring("""
<article>
<contrib-group>
<contrib>
<name>
<surname>Smith</surname>
<given-names>John</given-names>
</name>
</contrib>
</contrib-group>
<aff id="aff1">
<label>1</label>
<institution content-type="original">University A</institution>
</aff>
</article>
""")
result = xml_pipe.extract_contrib_data(xml)
self.assertEqual(result['affiliations'], ['1[^] University A'])


class TestExtractDOI(unittest.TestCase):

Expand Down