diff options
Diffstat (limited to 'backend/tolData/dbpedia')
| -rw-r--r-- | backend/tolData/dbpedia/README.md | 29 | ||||
| -rwxr-xr-x | backend/tolData/dbpedia/genDescData.py | 128 |
2 files changed, 0 insertions, 157 deletions
diff --git a/backend/tolData/dbpedia/README.md b/backend/tolData/dbpedia/README.md deleted file mode 100644 index dd9bda7..0000000 --- a/backend/tolData/dbpedia/README.md +++ /dev/null @@ -1,29 +0,0 @@ -This directory holds files obtained/derived from [Dbpedia](https://www.dbpedia.org). - -# Downloaded Files -- `labels_lang=en.ttl.bz2` <br> - Obtained via https://databus.dbpedia.org/dbpedia/collections/latest-core. - Downloaded from <https://databus.dbpedia.org/dbpedia/generic/labels/2022.03.01/labels_lang=en.ttl.bz2>. -- `page_lang=en_ids.ttl.bz2` <br> - Downloaded from <https://databus.dbpedia.org/dbpedia/generic/page/2022.03.01/page_lang=en_ids.ttl.bz2> -- `redirects_lang=en_transitive.ttl.bz2` <br> - Downloaded from <https://databus.dbpedia.org/dbpedia/generic/redirects/2022.03.01/redirects_lang=en_transitive.ttl.bz2>. -- `disambiguations_lang=en.ttl.bz2` <br> - Downloaded from <https://databus.dbpedia.org/dbpedia/generic/disambiguations/2022.03.01/disambiguations_lang=en.ttl.bz2>. -- `instance-types_lang=en_specific.ttl.bz2` <br> - Downloaded from <https://databus.dbpedia.org/dbpedia/mappings/instance-types/2022.03.01/instance-types_lang=en_specific.ttl.bz2>. -- `short-abstracts_lang=en.ttl.bz2` <br> - Downloaded from <https://databus.dbpedia.org/vehnem/text/short-abstracts/2021.05.01/short-abstracts_lang=en.ttl.bz2>. - -# Other Files -- genDescData.py <br> - Used to generate a database representing data from the ttl files. -- descData.db <br> - Generated by genDescData.py. <br> - Tables: <br> - - `labels`: `iri TEXT PRIMARY KEY, label TEXT ` - - `ids`: `iri TEXT PRIMARY KEY, id INT` - - `redirects`: `iri TEXT PRIMARY KEY, target TEXT` - - `disambiguations`: `iri TEXT PRIMARY KEY` - - `types`: `iri TEXT, type TEXT` - - `abstracts`: `iri TEXT PRIMARY KEY, abstract TEXT` diff --git a/backend/tolData/dbpedia/genDescData.py b/backend/tolData/dbpedia/genDescData.py deleted file mode 100755 index 43ed815..0000000 --- a/backend/tolData/dbpedia/genDescData.py +++ /dev/null @@ -1,128 +0,0 @@ -#!/usr/bin/python3 - -import re -import bz2, sqlite3 - -import argparse -parser = argparse.ArgumentParser(description=""" -Adds DBpedia labels/types/abstracts/etc data into a database -""", formatter_class=argparse.RawDescriptionHelpFormatter) -parser.parse_args() - -labelsFile = 'labels_lang=en.ttl.bz2' # Had about 16e6 entries -idsFile = 'page_lang=en_ids.ttl.bz2' -redirectsFile = 'redirects_lang=en_transitive.ttl.bz2' -disambigFile = 'disambiguations_lang=en.ttl.bz2' -typesFile = 'instance-types_lang=en_specific.ttl.bz2' -abstractsFile = 'short-abstracts_lang=en.ttl.bz2' -dbFile = 'descData.db' -# In testing, this script took a few hours to run, and generated about 10GB - -print('Creating database') -dbCon = sqlite3.connect(dbFile) -dbCur = dbCon.cursor() - -print('Reading/storing label data') -dbCur.execute('CREATE TABLE labels (iri TEXT PRIMARY KEY, label TEXT)') -dbCur.execute('CREATE INDEX labels_idx ON labels(label)') -dbCur.execute('CREATE INDEX labels_idx_nc ON labels(label COLLATE NOCASE)') -labelLineRegex = re.compile(r'<([^>]+)> <[^>]+> "((?:[^"]|\\")+)"@en \.\n') -lineNum = 0 -with bz2.open(labelsFile, mode='rt') as file: - for line in file: - lineNum += 1 - if lineNum % 1e5 == 0: - print(f'At line {lineNum}') - # - match = labelLineRegex.fullmatch(line) - if match is None: - raise Exception(f'ERROR: Line {lineNum} has unexpected format') - dbCur.execute('INSERT INTO labels VALUES (?, ?)', (match.group(1), match.group(2))) - -print('Reading/storing wiki page ids') -dbCur.execute('CREATE TABLE ids (iri TEXT PRIMARY KEY, id INT)') -dbCur.execute('CREATE INDEX ids_idx ON ids(id)') -idLineRegex = re.compile(r'<([^>]+)> <[^>]+> "(\d+)".*\n') -lineNum = 0 -with bz2.open(idsFile, mode='rt') as file: - for line in file: - lineNum += 1 - if lineNum % 1e5 == 0: - print(f'At line {lineNum}') - # - match = idLineRegex.fullmatch(line) - if match is None: - raise Exception(f'ERROR: Line {lineNum} has unexpected format') - try: - dbCur.execute('INSERT INTO ids VALUES (?, ?)', (match.group(1), int(match.group(2)))) - except sqlite3.IntegrityError as e: - # Accounts for certain lines that have the same IRI - print(f'WARNING: Failed to add entry with IRI "{match.group(1)}": {e}') - -print('Reading/storing redirection data') -dbCur.execute('CREATE TABLE redirects (iri TEXT PRIMARY KEY, target TEXT)') -redirLineRegex = re.compile(r'<([^>]+)> <[^>]+> <([^>]+)> \.\n') -lineNum = 0 -with bz2.open(redirectsFile, mode='rt') as file: - for line in file: - lineNum += 1 - if lineNum % 1e5 == 0: - print(f'At line {lineNum}') - # - match = redirLineRegex.fullmatch(line) - if match is None: - raise Exception(f'ERROR: Line {lineNum} has unexpected format') - dbCur.execute('INSERT INTO redirects VALUES (?, ?)', (match.group(1), match.group(2))) - -print('Reading/storing diambiguation-page data') -dbCur.execute('CREATE TABLE disambiguations (iri TEXT PRIMARY KEY)') -disambigLineRegex = redirLineRegex -lineNum = 0 -with bz2.open(disambigFile, mode='rt') as file: - for line in file: - lineNum += 1 - if lineNum % 1e5 == 0: - print(f'At line {lineNum}') - # - match = disambigLineRegex.fullmatch(line) - if match is None: - raise Exception(f'ERROR: Line {lineNum} has unexpected format') - dbCur.execute('INSERT OR IGNORE INTO disambiguations VALUES (?)', (match.group(1),)) - -print('Reading/storing instance-type data') -dbCur.execute('CREATE TABLE types (iri TEXT, type TEXT)') -dbCur.execute('CREATE INDEX types_iri_idx ON types(iri)') -typeLineRegex = redirLineRegex -lineNum = 0 -with bz2.open(typesFile, mode='rt') as file: - for line in file: - lineNum += 1 - if lineNum % 1e5 == 0: - print(f'At line {lineNum}') - # - match = typeLineRegex.fullmatch(line) - if match is None: - raise Exception(f'ERROR: Line {lineNum} has unexpected format') - dbCur.execute('INSERT INTO types VALUES (?, ?)', (match.group(1), match.group(2))) - -print('Reading/storing abstracts') -dbCur.execute('CREATE TABLE abstracts (iri TEXT PRIMARY KEY, abstract TEXT)') -descLineRegex = labelLineRegex -lineNum = 0 -with bz2.open(abstractsFile, mode='rt') as file: - for line in file: - lineNum += 1 - if lineNum % 1e5 == 0: - print(f'At line {lineNum}') - # - if line[0] == '#': - continue - match = descLineRegex.fullmatch(line) - if match is None: - raise Exception(f'ERROR: Line {lineNum} has unexpected format') - dbCur.execute('INSERT INTO abstracts VALUES (?, ?)', - (match.group(1), match.group(2).replace(r'\"', '"'))) - -print('Closing database') -dbCon.commit() -dbCon.close() |
