aboutsummaryrefslogtreecommitdiff
path: root/backend/tolData/dbpedia
diff options
context:
space:
mode:
authorTerry Truong <terry06890@gmail.com>2022-09-11 14:55:42 +1000
committerTerry Truong <terry06890@gmail.com>2022-09-11 15:04:14 +1000
commit5de5fb93e50fe9006221b30ac4a66f1be0db82e7 (patch)
tree2567c25c902dbb40d44419805cebb38171df47fa /backend/tolData/dbpedia
parentdaccbbd9c73a5292ea9d6746560d7009e5aa666d (diff)
Add backend unit tests
- Add unit testing code in backend/tests/ - Change to snake-case for script/file/directory names - Use os.path.join() instead of '/' - Refactor script code into function defs and a main-guard - Make global vars all-caps Some fixes: - For getting descriptions, some wiki redirects weren't properly resolved - Linked images were sub-optimally propagated - Generation of reduced trees assumed a wiki-id association implied a description - Tilo.py had potential null dereferences by not always using a reduced node set - EOL image downloading didn't properly wait for all threads to end when finishing
Diffstat (limited to 'backend/tolData/dbpedia')
-rw-r--r--backend/tolData/dbpedia/README.md29
-rwxr-xr-xbackend/tolData/dbpedia/genDescData.py128
2 files changed, 0 insertions, 157 deletions
diff --git a/backend/tolData/dbpedia/README.md b/backend/tolData/dbpedia/README.md
deleted file mode 100644
index dd9bda7..0000000
--- a/backend/tolData/dbpedia/README.md
+++ /dev/null
@@ -1,29 +0,0 @@
-This directory holds files obtained/derived from [Dbpedia](https://www.dbpedia.org).
-
-# Downloaded Files
-- `labels_lang=en.ttl.bz2` <br>
- Obtained via https://databus.dbpedia.org/dbpedia/collections/latest-core.
- Downloaded from <https://databus.dbpedia.org/dbpedia/generic/labels/2022.03.01/labels_lang=en.ttl.bz2>.
-- `page_lang=en_ids.ttl.bz2` <br>
- Downloaded from <https://databus.dbpedia.org/dbpedia/generic/page/2022.03.01/page_lang=en_ids.ttl.bz2>
-- `redirects_lang=en_transitive.ttl.bz2` <br>
- Downloaded from <https://databus.dbpedia.org/dbpedia/generic/redirects/2022.03.01/redirects_lang=en_transitive.ttl.bz2>.
-- `disambiguations_lang=en.ttl.bz2` <br>
- Downloaded from <https://databus.dbpedia.org/dbpedia/generic/disambiguations/2022.03.01/disambiguations_lang=en.ttl.bz2>.
-- `instance-types_lang=en_specific.ttl.bz2` <br>
- Downloaded from <https://databus.dbpedia.org/dbpedia/mappings/instance-types/2022.03.01/instance-types_lang=en_specific.ttl.bz2>.
-- `short-abstracts_lang=en.ttl.bz2` <br>
- Downloaded from <https://databus.dbpedia.org/vehnem/text/short-abstracts/2021.05.01/short-abstracts_lang=en.ttl.bz2>.
-
-# Other Files
-- genDescData.py <br>
- Used to generate a database representing data from the ttl files.
-- descData.db <br>
- Generated by genDescData.py. <br>
- Tables: <br>
- - `labels`: `iri TEXT PRIMARY KEY, label TEXT `
- - `ids`: `iri TEXT PRIMARY KEY, id INT`
- - `redirects`: `iri TEXT PRIMARY KEY, target TEXT`
- - `disambiguations`: `iri TEXT PRIMARY KEY`
- - `types`: `iri TEXT, type TEXT`
- - `abstracts`: `iri TEXT PRIMARY KEY, abstract TEXT`
diff --git a/backend/tolData/dbpedia/genDescData.py b/backend/tolData/dbpedia/genDescData.py
deleted file mode 100755
index 43ed815..0000000
--- a/backend/tolData/dbpedia/genDescData.py
+++ /dev/null
@@ -1,128 +0,0 @@
-#!/usr/bin/python3
-
-import re
-import bz2, sqlite3
-
-import argparse
-parser = argparse.ArgumentParser(description="""
-Adds DBpedia labels/types/abstracts/etc data into a database
-""", formatter_class=argparse.RawDescriptionHelpFormatter)
-parser.parse_args()
-
-labelsFile = 'labels_lang=en.ttl.bz2' # Had about 16e6 entries
-idsFile = 'page_lang=en_ids.ttl.bz2'
-redirectsFile = 'redirects_lang=en_transitive.ttl.bz2'
-disambigFile = 'disambiguations_lang=en.ttl.bz2'
-typesFile = 'instance-types_lang=en_specific.ttl.bz2'
-abstractsFile = 'short-abstracts_lang=en.ttl.bz2'
-dbFile = 'descData.db'
-# In testing, this script took a few hours to run, and generated about 10GB
-
-print('Creating database')
-dbCon = sqlite3.connect(dbFile)
-dbCur = dbCon.cursor()
-
-print('Reading/storing label data')
-dbCur.execute('CREATE TABLE labels (iri TEXT PRIMARY KEY, label TEXT)')
-dbCur.execute('CREATE INDEX labels_idx ON labels(label)')
-dbCur.execute('CREATE INDEX labels_idx_nc ON labels(label COLLATE NOCASE)')
-labelLineRegex = re.compile(r'<([^>]+)> <[^>]+> "((?:[^"]|\\")+)"@en \.\n')
-lineNum = 0
-with bz2.open(labelsFile, mode='rt') as file:
- for line in file:
- lineNum += 1
- if lineNum % 1e5 == 0:
- print(f'At line {lineNum}')
- #
- match = labelLineRegex.fullmatch(line)
- if match is None:
- raise Exception(f'ERROR: Line {lineNum} has unexpected format')
- dbCur.execute('INSERT INTO labels VALUES (?, ?)', (match.group(1), match.group(2)))
-
-print('Reading/storing wiki page ids')
-dbCur.execute('CREATE TABLE ids (iri TEXT PRIMARY KEY, id INT)')
-dbCur.execute('CREATE INDEX ids_idx ON ids(id)')
-idLineRegex = re.compile(r'<([^>]+)> <[^>]+> "(\d+)".*\n')
-lineNum = 0
-with bz2.open(idsFile, mode='rt') as file:
- for line in file:
- lineNum += 1
- if lineNum % 1e5 == 0:
- print(f'At line {lineNum}')
- #
- match = idLineRegex.fullmatch(line)
- if match is None:
- raise Exception(f'ERROR: Line {lineNum} has unexpected format')
- try:
- dbCur.execute('INSERT INTO ids VALUES (?, ?)', (match.group(1), int(match.group(2))))
- except sqlite3.IntegrityError as e:
- # Accounts for certain lines that have the same IRI
- print(f'WARNING: Failed to add entry with IRI "{match.group(1)}": {e}')
-
-print('Reading/storing redirection data')
-dbCur.execute('CREATE TABLE redirects (iri TEXT PRIMARY KEY, target TEXT)')
-redirLineRegex = re.compile(r'<([^>]+)> <[^>]+> <([^>]+)> \.\n')
-lineNum = 0
-with bz2.open(redirectsFile, mode='rt') as file:
- for line in file:
- lineNum += 1
- if lineNum % 1e5 == 0:
- print(f'At line {lineNum}')
- #
- match = redirLineRegex.fullmatch(line)
- if match is None:
- raise Exception(f'ERROR: Line {lineNum} has unexpected format')
- dbCur.execute('INSERT INTO redirects VALUES (?, ?)', (match.group(1), match.group(2)))
-
-print('Reading/storing diambiguation-page data')
-dbCur.execute('CREATE TABLE disambiguations (iri TEXT PRIMARY KEY)')
-disambigLineRegex = redirLineRegex
-lineNum = 0
-with bz2.open(disambigFile, mode='rt') as file:
- for line in file:
- lineNum += 1
- if lineNum % 1e5 == 0:
- print(f'At line {lineNum}')
- #
- match = disambigLineRegex.fullmatch(line)
- if match is None:
- raise Exception(f'ERROR: Line {lineNum} has unexpected format')
- dbCur.execute('INSERT OR IGNORE INTO disambiguations VALUES (?)', (match.group(1),))
-
-print('Reading/storing instance-type data')
-dbCur.execute('CREATE TABLE types (iri TEXT, type TEXT)')
-dbCur.execute('CREATE INDEX types_iri_idx ON types(iri)')
-typeLineRegex = redirLineRegex
-lineNum = 0
-with bz2.open(typesFile, mode='rt') as file:
- for line in file:
- lineNum += 1
- if lineNum % 1e5 == 0:
- print(f'At line {lineNum}')
- #
- match = typeLineRegex.fullmatch(line)
- if match is None:
- raise Exception(f'ERROR: Line {lineNum} has unexpected format')
- dbCur.execute('INSERT INTO types VALUES (?, ?)', (match.group(1), match.group(2)))
-
-print('Reading/storing abstracts')
-dbCur.execute('CREATE TABLE abstracts (iri TEXT PRIMARY KEY, abstract TEXT)')
-descLineRegex = labelLineRegex
-lineNum = 0
-with bz2.open(abstractsFile, mode='rt') as file:
- for line in file:
- lineNum += 1
- if lineNum % 1e5 == 0:
- print(f'At line {lineNum}')
- #
- if line[0] == '#':
- continue
- match = descLineRegex.fullmatch(line)
- if match is None:
- raise Exception(f'ERROR: Line {lineNum} has unexpected format')
- dbCur.execute('INSERT INTO abstracts VALUES (?, ?)',
- (match.group(1), match.group(2).replace(r'\"', '"')))
-
-print('Closing database')
-dbCon.commit()
-dbCon.close()