Initial commit

This commit is contained in:
Luka Romih
2022-12-07 04:43:36 +01:00
commit 257f3c354f
496 changed files with 55373 additions and 0 deletions
@@ -0,0 +1,282 @@
const fs = require('fs')
const xmlFlow = require('xml-flow')
const xss = require('xss')
const debug = require('debug')('termPortal:models/helpers/dictionary')
const db = require('../../db')
const { intoDbArray } = require('..')
const JOBS_MAX = 50
const JOBS_MIN = 15
// Import given file into given dictionary.
exports.readFileIntoDb = (
userId,
dictionaryId,
importFilePath,
entryStatus,
dbClient
) => {
return new Promise((resolve, reject) => {
const progress = {
jobCount: 0,
totalCount: 0,
validCount: 0,
isWholeFileRead: false,
importError: null
}
entryStatus = entryStatus === 'complete' ? 'complete' : 'in_edit'
const importFileReader = fs.createReadStream(importFilePath)
const xmlStream = xmlFlow(importFileReader, {
strict: true,
trim: false,
preserveMarkup: xmlFlow.ALWAYS,
simplifyNodes: false,
useArrays: xmlFlow.ALWAYS
})
const handleError = handleImportError(xmlStream, progress, reject)
const handleEntry = handleEntryXml(
userId,
dictionaryId,
entryStatus,
progress,
dbClient,
resolve,
handleError
)
xmlStream
.on('error', () => {})
.once('error', handleError)
.on('end', () => (progress.isWholeFileRead = true))
.on('tag:entry', handleEntry)
})
}
function handleImportError(xmlStream, progress, reject) {
return function handleError(error) {
if (progress.importError) return
progress.importError = error
xmlStream.removeAllListeners('tag:entry')
reject(error)
}
}
function handleEntryXml(
userId,
dictionaryId,
entryStatus,
progress,
dbClient,
resolve,
handleError
) {
return async function handleEntry(entry) {
if (progress.importError) return
try {
if (++progress.jobCount > JOBS_MAX) this.pause()
const term = toMixedBasic(
entry.$markup.find(el => el.$name === 'term')?.$markup
)
const hwGrp = entry.$markup.find(el => el.$name === 'hwGrp')
const wordforms = hwGrp?.$attrs.wfs
const accent = hwGrp?.$attrs.acc
const pronunciation = hwGrp?.$attrs.pron
const domainLabels = entry.$markup
.find(el => el.$name === 'domainLabels')
?.$markup.filter(childEl => childEl.$name === 'domainLabel')
.map(domainLabel => toText(domainLabel.$markup))
const label = toMixedExtended(
entry.$markup.find(el => el.$name === 'label')?.$markup
)
const definition = toMixedExtended(
entry.$markup.find(el => el.$name === 'def')?.$markup
)
const synonyms = entry.$markup
.find(el => el.$name === 'syns')
?.$markup.filter(childEl => childEl.$name === 'syn')
.map(synonym => toMixedBasic(synonym.$markup))
const links = entry.$markup
.find(el => el.$name === 'links')
?.$markup.filter(childEl => childEl.$name === 'link')
.map(link => ({
link: toMixedBasic(link.$markup),
type: link.$attrs.type
}))
const other = toMixedOther(
entry.$markup.find(el => el.$name === 'other')?.$markup
)
let hasForeignTerms = false
const foreignLanguageContent = entry.$markup
.find(el => el.$name === 'fLangs')
?.$markup.filter(childEl => childEl.$name === 'fLang')
.map(fLang => ({
language: fLang.$attrs.lang,
terms: fLang.$markup
.find(el => el.$name === 'fTerms')
?.$markup.filter(childEl => childEl.$name === 'fTerm')
.map(term => {
const content = toMixedBasic(term.$markup)
if (content.length) hasForeignTerms = true
return content
}),
definition:
toMixedExtended(
fLang.$markup.find(el => el.$name === 'fDef')?.$markup
) || null,
synonyms: fLang.$markup
.find(el => el.$name === 'fSyns')
?.$markup.filter(childEl => childEl.$name === 'fSyn')
.map(synonym => toMixedBasic(synonym.$markup))
}))
const multimedia = entry.$markup.find(el => el.$name === 'mm')?.$markup
const images = []
const audio = []
const videos = []
multimedia?.forEach(mm => {
switch (mm.$name) {
case 'image':
images.push(toText(mm.$markup))
break
case 'audio':
audio.push(toText(mm.$markup))
break
case 'video':
videos.push(toText(mm.$markup))
}
})
const isValid = !!term && (!!definition || hasForeignTerms)
if (!entry.$markup.length) return
const values = [
dictionaryId,
isValid,
entryStatus,
term || null,
userId,
null,
wordforms,
accent,
pronunciation,
intoDbArray(domainLabels, 'always'),
label || null,
definition || null,
intoDbArray(synonyms),
intoDbArray(links, 'always'),
other || null,
intoDbArray(foreignLanguageContent, 'always'),
images.length ? images : null,
audio.length ? audio : null,
videos.length ? videos : null
]
const text = `SELECT entry_new (${db.genParamStr(values)})`
await dbClient.query(text, values)
progress.totalCount++
if (isValid) progress.validCount++
} catch (error) {
handleError(error)
} finally {
progress.jobCount--
if (progress.jobCount < JOBS_MIN) this.resume()
if (
!progress.jobCount &&
progress.isWholeFileRead &&
!progress.importError
) {
debug(`TOTAL ENTRIES: ${progress.totalCount}`)
debug(`VALID ENTRIES: ${progress.validCount}`)
resolve()
}
}
}
}
const markupFilter = {
noMixed: new xss.FilterXSS({
whiteList: {},
stripIgnoreTag: true,
stripIgnoreTagBody: ['script', 'style']
}),
mixedBasic: new xss.FilterXSS({
whiteList: {
sup: [],
sub: []
},
stripIgnoreTag: true,
stripIgnoreTagBody: ['script', 'style']
}),
mixedExtended: new xss.FilterXSS({
whiteList: {
sup: [],
sub: [],
b: [],
i: [],
a: ['href']
},
stripIgnoreTag: true,
stripIgnoreTagBody: ['script', 'style'],
onTag: customTagHandler
}),
mixedOther: new xss.FilterXSS({
whiteList: {
sup: [],
sub: [],
b: [],
i: [],
a: ['href'],
br: []
},
stripIgnoreTag: true,
stripIgnoreTagBody: ['script', 'style'],
onTag: customTagHandler
})
}
function customTagHandler(tag, html, { isWhite, isClosing }) {
// Special treatment only for whitelisted opening anchor tags.
if (tag !== 'a' || !isWhite || isClosing) return
const matchUrl = html.match(/href="?(?<url>https?:\/\/.*?)"?[\s>]/)
const url = matchUrl ? xss.escapeAttrValue(matchUrl.groups.url) : undefined
return `<a href${url ? `="${url}" target="_blank"` : ''}>`
}
function toText(markupObj) {
return markupFilter.noMixed
.process(xmlFlow.toXml(markupObj))
.replace(/\s+/g, ' ')
.trim()
}
function toMixedBasic(markupObj) {
return markupFilter.mixedBasic
.process(xmlFlow.toXml(markupObj))
.replace(/\s+/g, ' ')
.trim()
}
function toMixedExtended(markupObj) {
return markupFilter.mixedExtended
.process(xmlFlow.toXml(markupObj))
.replace(/\s+/g, ' ')
.trim()
}
function toMixedOther(markupObj) {
return markupFilter.mixedOther
.process(xmlFlow.toXml(markupObj))
.replace(/\s+/g, ' ')
.trim()
}
+196
View File
@@ -0,0 +1,196 @@
const { removeHtmlTags } = require('../../helpers')
const { searchEngineClient, ENTRY_INDEX } = require('../../search-engine')
exports.deserialize = {
primaryDomain(domain) {
const deserializedDomain = {
id: domain.id,
nameSl: domain.name_sl,
nameEn: domain.name_en
}
return deserializedDomain
},
secondaryDomain(domain) {
const deserializedDomain = {
id: domain.id,
isApproved: domain.approved,
nameSl: domain.name_sl,
nameEn: domain.name_en
}
return deserializedDomain
},
approvedSecondaryDomain(domain) {
const deserializedDomain = {
id: domain.id,
nameSl: domain.name_sl,
nameEn: domain.name_en
}
return deserializedDomain
},
language(language) {
const deserializedLanguage = {
id: language.id,
code: language.code,
nameSl: language.name_sl,
nameEn: language.name_en
}
return deserializedLanguage
},
editDescription(dictionary) {
const deserializedDictionary = {
id: dictionary.id,
nameSl: dictionary.name_sl,
nameEn: dictionary.name_en,
nameSlShort: dictionary.name_sl_short,
author: dictionary.author,
domainPrimary: dictionary.domain_primary_id,
description: dictionary.description,
issn: dictionary.issn
}
return deserializedDictionary
},
editUsers(dictionary) {
const deserializedDictionary = {
id: dictionary.id,
nameSl: dictionary.name_sl,
terminologyReviewFlag: dictionary.entries_have_terminology_review_flag,
languageReviewFlag: dictionary.entries_have_language_review_flag,
status: dictionary.status
}
return deserializedDictionary
},
editStructure(dictionary) {
const deserializedDictionary = {
id: dictionary.id,
nameSl: dictionary.name_sl,
hasDomainLabels: dictionary.entries_have_domain_labels,
hasLabel: dictionary.entries_have_label,
hasDefinition: dictionary.entries_have_definition,
hasSynonyms: dictionary.entries_have_synonyms,
hasLinks: dictionary.entries_have_links,
hasOther: dictionary.entries_have_other,
hasForeignLanguages: dictionary.entries_have_foreign_languages,
hasForeignDefinitions: dictionary.entries_have_foreign_definitions,
hasForeignSynonyms: dictionary.entries_have_foreign_synonyms,
hasImages: dictionary.entries_have_images,
hasAudio: dictionary.entries_have_audio,
hasVideo: dictionary.entries_have_videos
}
return deserializedDictionary
},
editDomainLabels(domainLabel) {
const deserializedDomainLabel = {
id: domainLabel.id,
name: domainLabel.name,
isVisible: domainLabel.is_visible
}
return deserializedDomainLabel
},
imports(oneImport) {
const deserializedImports = {
timeStarted: oneImport.time_started,
status: oneImport.status,
deleteExisting: oneImport.delete_existing_entries,
fileFormat: oneImport.file_format,
countValidEntries: oneImport.count_valid_entries
}
return deserializedImports
}
}
exports.bulkIndex = async (entries, primaryDomain, dictionary, source) => {
const entriesCount = entries.length
if (!entries.length) return
// Construct request body.
const bulkBody = new Array(entriesCount * 2)
for (let i = 0; i < entriesCount; i++) {
let entry = prepareEntryForIndexing(entries[i])
bulkBody[i * 2] = { index: { _index: ENTRY_INDEX, _id: entry.id } }
entry.primaryDomain = primaryDomain
entry.dictionary = dictionary
entry.source = source
entry = removeHtmlTags(JSON.stringify(entry))
bulkBody[i * 2 + 1] = entry
}
// Send the request.
const bulkResponse = await searchEngineClient.bulk({ body: bulkBody })
// Log possible errors.
if (bulkResponse.errors) {
const erroredDocuments = []
bulkResponse.items.forEach((action, i) => {
const operation = Object.keys(action)[0]
if (action[operation].error) {
erroredDocuments.push({
// If the status is 429 it means that you can retry the document,
// otherwise it's very likely a mapping error, and you should
// fix the document before to try it again.
status: action[operation].status,
error: action[operation].error,
operation: action[operation].body[i * 2],
document: action[operation].body[i * 2 + 1]
})
}
})
console.log('Errors indexing documents:') // eslint-disable-line no-console
console.log(erroredDocuments) // eslint-disable-line no-console
}
}
function prepareEntryForIndexing(entry) {
// Snake case property names into camel case.
entry.isValid = entry.is_valid
delete entry.is_valid
entry.isPublished = entry.is_published
delete entry.is_published
entry.isTerminologyReviewed = entry.is_terminology_reviewed
delete entry.is_terminology_reviewed
entry.isLanguageReviewed = entry.is_language_reviewed
delete entry.is_language_reviewed
entry.homonymSort = entry.homonym_sort
delete entry.homonym_sort
entry.timeMostRecentComment = entry.time_most_recent_comment
delete entry.time_most_recent_comment
entry.domainLabels = entry.domain_labels
delete entry.domain_labels
entry.foreignEntries = entry.foreign_entries
delete entry.foreign_entries
// Remove (top-level) properies with null or empty array values.
entry = Object.fromEntries(
Object.entries(entry).filter(
([_, v]) => v !== null && (!Array.isArray(v) || v.length)
)
)
return entry
}
exports.prepareEntryForIndexing = prepareEntryForIndexing