aboutsummaryrefslogtreecommitdiffstats
path: root/python
diff options
context:
space:
mode:
authorBryan Newbold <bnewbold@robocracy.org>2019-12-23 17:59:34 -0800
committerBryan Newbold <bnewbold@robocracy.org>2019-12-23 18:18:26 -0800
commit87673154a1b2f51b0bb24e1f83d608ea57ff5936 (patch)
treec442fc43478ef653b9f7822ac64a9d2b0bba04ca /python
parent95c8a10d8bd424ac40267bf9677c2a7d776c828b (diff)
downloadfatcat-87673154a1b2f51b0bb24e1f83d608ea57ff5936.tar.gz
fatcat-87673154a1b2f51b0bb24e1f83d608ea57ff5936.zip
normalizers: clean_pmid(), and handle nulls in all other cleaners
Diffstat (limited to 'python')
-rw-r--r--python/fatcat_tools/normal.py31
1 files changed, 31 insertions, 0 deletions
diff --git a/python/fatcat_tools/normal.py b/python/fatcat_tools/normal.py
index 80bcfa5a..1cc66633 100644
--- a/python/fatcat_tools/normal.py
+++ b/python/fatcat_tools/normal.py
@@ -19,6 +19,8 @@ def clean_doi(raw):
Returns None if not a valid DOI
"""
+ if not raw:
+ return None
raw = raw.strip()
if len(raw.split()) != 1:
return None
@@ -54,6 +56,8 @@ def clean_arxiv_id(raw):
Works with versioned or un-versioned arxiv identifiers.
"""
+ if not raw:
+ return None
raw = raw.strip()
if raw.lower().startswith("arxiv:"):
raw = raw[6:]
@@ -90,7 +94,26 @@ def test_clean_arxiv_id():
assert clean_arxiv_id("0806.v1") == None
assert clean_arxiv_id("08062878v1") == None
+def clean_pmid(raw):
+ if not raw:
+ return None
+ raw = raw.strip()
+ if len(raw.split()) != 1:
+ return None
+ if raw.isdigit():
+ return raw
+ return None
+
+def test_clean_pmid():
+ assert clean_pmid("1234") == "1234"
+ assert clean_pmid("1234 ") == "1234"
+ assert clean_pmid("PMC123") == None
+ assert clean_sha1("qfba3") == None
+ assert clean_sha1("") == None
+
def clean_pmcid(raw):
+ if not raw:
+ return None
raw = raw.strip()
if len(raw.split()) != 1:
return None
@@ -99,6 +122,8 @@ def clean_pmcid(raw):
return None
def clean_sha1(raw):
+ if not raw:
+ return None
raw = raw.strip().lower()
if len(raw.split()) != 1:
return None
@@ -134,6 +159,8 @@ def test_clean_sha256():
ISSN_REGEX = re.compile("^\d{4}-\d{3}[0-9X]$")
def clean_issn(raw):
+ if not raw:
+ return None
raw = raw.strip().upper()
if len(raw) != 9:
return None
@@ -150,6 +177,8 @@ def test_clean_issn():
ISBN13_REGEX = re.compile("^97(?:8|9)-\d{1,5}-\d{1,7}-\d{1,6}-\d$")
def clean_isbn13(raw):
+ if not raw:
+ return None
raw = raw.strip()
if not ISBN13_REGEX.fullmatch(raw):
return None
@@ -164,6 +193,8 @@ def test_clean_isbn13():
ORCID_REGEX = re.compile("^\d{4}-\d{4}-\d{4}-\d{3}[\dX]$")
def clean_orcid(raw):
+ if not raw:
+ return None
raw = raw.strip()
if not ORCID_REGEX.fullmatch(raw):
return None