diff --git a/services/irc3-species/Dockerfile b/services/irc3-species/Dockerfile index 341c7d1e5..7a56b7a04 100644 --- a/services/irc3-species/Dockerfile +++ b/services/irc3-species/Dockerfile @@ -1,5 +1,5 @@ # syntax=docker/dockerfile:1.2 -FROM cnrsinist/ezs-python-server:py3.9-no24-1.0.14 +FROM cnrsinist/ezs-python-server:py3.10-no24-2.0.1 WORKDIR /app/public diff --git a/services/irc3-species/README.md b/services/irc3-species/README.md index a87f950e4..b28cf9ad5 100644 --- a/services/irc3-species/README.md +++ b/services/irc3-species/README.md @@ -1,4 +1,4 @@ -# ws-irc3-species@1.1.12 +# ws-irc3-species@1.1.13 IRC3 dédiée à la recherche des noms scientifiques diff --git a/services/irc3-species/config.json b/services/irc3-species/config.json index 5cab609ca..cf620f623 100644 --- a/services/irc3-species/config.json +++ b/services/irc3-species/config.json @@ -11,7 +11,7 @@ "IRC3SP_DEBUG": false, "IRC3SP_LOG": false, "IRC3SP_TABLE": "/app/public/v1/CoL.txt", - "IRC3SP_CASE_SENSITIVE": false, + "IRC3SP_CASE_SENSITIVE": true, "NODE_OPTIONS": "--max_old_space_size=1024", "NODE_ENV": "production" } diff --git a/services/irc3-species/examples.http b/services/irc3-species/examples.http index ef5b6f722..fba644e6b 100644 --- a/services/irc3-species/examples.http +++ b/services/irc3-species/examples.http @@ -35,5 +35,41 @@ Content-Type: application/json { "id": 6, "value": "Atmospheric Dioxin and Furan Deposition in Relation to Land-Use and Other Pollutants: A Survey with Lichens>Abstract: Polychlorodibenzeno-dioxins and polychlorodibenzeno-furans (PCDD/Fs) are considered among the most toxic compounds on earth. The aim of the present study was to evaluate atmospheric PCDD/F deposition and identify the areas with greater deposition of these compounds in an important industrialized and urbanized region of Portugal, using lichens as biomonitors. For this purpose, samples of the lichen Xanthoria parietina were collected at 60 sampling sites, covering urban, industrial, forestry and agriculture areas, and analyzed for PCDD/Fs, sulfur, nitrogen, zinc, iron, chromium, lead, cobalt, nickel, copper, calcium, manganese, magnesium and potassium. The concentrations of PCDD/Fs in lichens were compared with the other elements and related to land-use and population density. The results obtained through the geostatistical interpolations and after principal component analysis have shown that PCDD/F deposition estimated by lichens is greater near industrial and highly populated urban areas. We found that lichens are suitable biomonitors of PCDD/F atmospheric deposition and can contribute to a better knowledge of air quality in a region, enabling identification of critical pollutant deposition areas." + }, + { + "id": 7, + "value": "Comparison of the air pollution biomonitoring ability of three Tillandsia species and the lichen Ramalina celastri in Argentina>Abstract: Bioaccumulation ability and response to air pollution sources were evaluated for Tillandsia capillaris Ruíz and Pav. f. capillaris, T. recurvata L., T. tricholepis Baker and the lichen Ramalina celastri (Spreng.) Krog. and Swinsc." + }, + { + "id": 8, + "value": "Plants as bioindicators of air pollution at the serra do mar near the industrial complex of Cubatão, Brazil>Abstract: As a result of air pollutant emissions from the industrial complex of Cubatão, Brazil, the Atlantic Forest vegetation of the Serra do Mar shows severe and widespread damage. In order to obtain information on the type, intensity and causes of the vegetation damage, bioindicator plants were exposed at different distances from the emission sources. Air-pollution-induced effects were evaluated by estimation of visible injury symptoms and chemical analyses of leaves. The results prove the occurrence of phytotoxic levels of photochemical oxidants in wide parts of the research area. Intense fluoride-induced damage and high leaf fluoride concentrations were found in a valley downwind of fertiliser industries. The study showed that some of the traditional standardised bioindication methods from temperate climates may be successfully employed in biomonitoring programmes in tropical and subtropical regions." + }, + { + "id": 9, + "value": "Morphometric differences of Microgramma squamulosa (Kaulf.) de la Sota (Polypodiaceae) leaves in environments with distinct atmospheric air quality>Plants growing in environments with different atmospheric conditions may present changes in the morphometric parameters of their leaves.Microgramma squamulosa (Kaulf.)de la Sota is a neotropical epiphytic fern found in impacted environments.The aims of this study were to quantitatively compare structural characteristics of leaves in areas with different air quality conditions, and to identify morphometric parameters that are potential indicators of the effects of pollution on these plants.Fertile and sterile leaves growing on isolated trees were collected from an urban (Estância Velha) and a rural (Novo Hamburgo) environment, in Rio Grande do Sul, Brazil.For each leaf type, macroscopic and microscopic analyses were performed on 192 samples collected in each environment.The sterile and fertile leaves showed significantly greater thickness of the midrib and greater vascular bundle and leaf blade areas in the rural environment, which is characterized by less air pollution.The thickness of the hypodermis and the stomatal density of the fertile leaves were greater in the urban area, which is characterized by more air pollution.Based on the fact that significant changes were found in the parameters of both types of leaves, which could possibly be related to air pollutants, M. squamulosa may be a potential bioindicator." + }, + { + "id": 10, + "value": "A Preliminary Assessment of the Montréal Process Indicators of Air Pollution for the United States>Abstract: Air pollutants pose a risk to forest health and vitality in the UnitedStates. Here we present the major findings from a national scale air pollution assessment that is part of the United States' 2003 Report on Sustainable Forests. We examine trends and the percent forest subjected to specific levels of ozone and wet deposition of sulfate, nitrate, and ammonium. Results are reported by Resource Planning Act (RPA) reporting region and integrated by forest type using multivariate clustering. Estimates of sulfate deposition for forested areas had decreasing trends (1994–2000) across RPA regions that were statistically significant for North and South RPA regions. Nitrate deposition rates were relatively constant for the 1994 to 2000 period, but the South RPA region had a statistically decreasing trend. The North and South RPA regions experienced the highest ammonium deposition rates and showed slightly decreasing trends. Ozone concentrations were highest in portions of the Pacific Coast RPA region and relatively high across much of the South RPA region. Both the South and Rocky Mountain RPA regions had an increasing trend in ozone exposure. Ozone-induced foliar injury to sensitive species was recorded in all regions except for the Rocky Mountain region. The multivariate analysis showed that the oak-hickory and loblolly-shortleaf pine forest types were generally exposed to more air pollution than other forest types, and the redwood, western white pine, and larch forest types were generally exposed to less. These findings offer a new approach to national air pollution assessments and are intended to help focus research and planning initiatives related to air pollution and forest health." + }, + { + "id": 11, + "value": "The lichen Usnea hirta is a bioindicator of air pollution, especially of sulphur dioxide." + }, + { + "id": 12, + "value": "In this study, the essential oil of Citrus aurantium L. peel was analysed by gas chromatography." + }, + { + "id": 13, + "value": "Samples of Mus musculus were collected in the industrial area to evaluate its use as a bioindicator." + }, + { + "id": 14, + "value": "Plants of the genus Tillandsia were studied, and specimens of T. recurvata L. and T. tricholepis Baker were collected." + }, + { + "id": 15, + "value": "The specimens of T. recurvata L. were collected near the factory." } ] diff --git a/services/irc3-species/package.json b/services/irc3-species/package.json index d5ff6854f..d9b2d9616 100644 --- a/services/irc3-species/package.json +++ b/services/irc3-species/package.json @@ -1,7 +1,7 @@ { "private": true, "name": "ws-irc3-species", - "version": "1.1.12", + "version": "1.1.13", "description": "IRC3 dédiée à la recherche des noms scientifiques", "repository": { "type": "git", diff --git a/services/irc3-species/swagger.json b/services/irc3-species/swagger.json index c88db490c..6ba15dc7c 100644 --- a/services/irc3-species/swagger.json +++ b/services/irc3-species/swagger.json @@ -3,7 +3,7 @@ "info": { "title": "irc3-species - IRC3 dédiée à la recherche des noms scientifiques", "description": "IRC3sp est une version de l’outil IRC3 dédiée à la recherche des noms scientifiques — ou noms binominaux — d’espèces animales, végétales ou autres dans un corpus de textes en se référant à une liste finie (mais, aussi exhaustive que possible).", - "version": "1.1.12", + "version": "1.1.13", "termsOfService": "https://services.istex.fr/", "contact": { "name": "Inist-CNRS", diff --git a/services/irc3-species/tests.hurl b/services/irc3-species/tests.hurl index 3ec61e267..9060cb824 100644 --- a/services/irc3-species/tests.hurl +++ b/services/irc3-species/tests.hurl @@ -24,6 +24,42 @@ content-type: application/json { "id": 6, "value": "Atmospheric Dioxin and Furan Deposition in Relation to Land-Use and Other Pollutants: A Survey with Lichens>Abstract: Polychlorodibenzeno-dioxins and polychlorodibenzeno-furans (PCDD/Fs) are considered among the most toxic compounds on earth. The aim of the present study was to evaluate atmospheric PCDD/F deposition and identify the areas with greater deposition of these compounds in an important industrialized and urbanized region of Portugal, using lichens as biomonitors. For this purpose, samples of the lichen Xanthoria parietina were collected at 60 sampling sites, covering urban, industrial, forestry and agriculture areas, and analyzed for PCDD/Fs, sulfur, nitrogen, zinc, iron, chromium, lead, cobalt, nickel, copper, calcium, manganese, magnesium and potassium. The concentrations of PCDD/Fs in lichens were compared with the other elements and related to land-use and population density. The results obtained through the geostatistical interpolations and after principal component analysis have shown that PCDD/F deposition estimated by lichens is greater near industrial and highly populated urban areas. We found that lichens are suitable biomonitors of PCDD/F atmospheric deposition and can contribute to a better knowledge of air quality in a region, enabling identification of critical pollutant deposition areas." + }, + { + "id": 7, + "value": "Comparison of the air pollution biomonitoring ability of three Tillandsia species and the lichen Ramalina celastri in Argentina>Abstract: Bioaccumulation ability and response to air pollution sources were evaluated for Tillandsia capillaris Ruíz and Pav. f. capillaris, T. recurvata L., T. tricholepis Baker and the lichen Ramalina celastri (Spreng.) Krog. and Swinsc." + }, + { + "id": 8, + "value": "Plants as bioindicators of air pollution at the serra do mar near the industrial complex of Cubatão, Brazil>Abstract: As a result of air pollutant emissions from the industrial complex of Cubatão, Brazil, the Atlantic Forest vegetation of the Serra do Mar shows severe and widespread damage. In order to obtain information on the type, intensity and causes of the vegetation damage, bioindicator plants were exposed at different distances from the emission sources. Air-pollution-induced effects were evaluated by estimation of visible injury symptoms and chemical analyses of leaves. The results prove the occurrence of phytotoxic levels of photochemical oxidants in wide parts of the research area. Intense fluoride-induced damage and high leaf fluoride concentrations were found in a valley downwind of fertiliser industries. The study showed that some of the traditional standardised bioindication methods from temperate climates may be successfully employed in biomonitoring programmes in tropical and subtropical regions." + }, + { + "id": 9, + "value": "Morphometric differences of Microgramma squamulosa (Kaulf.) de la Sota (Polypodiaceae) leaves in environments with distinct atmospheric air quality>Plants growing in environments with different atmospheric conditions may present changes in the morphometric parameters of their leaves.Microgramma squamulosa (Kaulf.)de la Sota is a neotropical epiphytic fern found in impacted environments.The aims of this study were to quantitatively compare structural characteristics of leaves in areas with different air quality conditions, and to identify morphometric parameters that are potential indicators of the effects of pollution on these plants.Fertile and sterile leaves growing on isolated trees were collected from an urban (Estância Velha) and a rural (Novo Hamburgo) environment, in Rio Grande do Sul, Brazil.For each leaf type, macroscopic and microscopic analyses were performed on 192 samples collected in each environment.The sterile and fertile leaves showed significantly greater thickness of the midrib and greater vascular bundle and leaf blade areas in the rural environment, which is characterized by less air pollution.The thickness of the hypodermis and the stomatal density of the fertile leaves were greater in the urban area, which is characterized by more air pollution.Based on the fact that significant changes were found in the parameters of both types of leaves, which could possibly be related to air pollutants, M. squamulosa may be a potential bioindicator." + }, + { + "id": 10, + "value": "A Preliminary Assessment of the Montréal Process Indicators of Air Pollution for the United States>Abstract: Air pollutants pose a risk to forest health and vitality in the UnitedStates. Here we present the major findings from a national scale air pollution assessment that is part of the United States' 2003 Report on Sustainable Forests. We examine trends and the percent forest subjected to specific levels of ozone and wet deposition of sulfate, nitrate, and ammonium. Results are reported by Resource Planning Act (RPA) reporting region and integrated by forest type using multivariate clustering. Estimates of sulfate deposition for forested areas had decreasing trends (1994–2000) across RPA regions that were statistically significant for North and South RPA regions. Nitrate deposition rates were relatively constant for the 1994 to 2000 period, but the South RPA region had a statistically decreasing trend. The North and South RPA regions experienced the highest ammonium deposition rates and showed slightly decreasing trends. Ozone concentrations were highest in portions of the Pacific Coast RPA region and relatively high across much of the South RPA region. Both the South and Rocky Mountain RPA regions had an increasing trend in ozone exposure. Ozone-induced foliar injury to sensitive species was recorded in all regions except for the Rocky Mountain region. The multivariate analysis showed that the oak-hickory and loblolly-shortleaf pine forest types were generally exposed to more air pollution than other forest types, and the redwood, western white pine, and larch forest types were generally exposed to less. These findings offer a new approach to national air pollution assessments and are intended to help focus research and planning initiatives related to air pollution and forest health." + }, + { + "id": 11, + "value": "The lichen Usnea hirta is a bioindicator of air pollution, especially of sulphur dioxide." + }, + { + "id": 12, + "value": "In this study, the essential oil of Citrus aurantium L. peel was analysed by gas chromatography." + }, + { + "id": 13, + "value": "Samples of Mus musculus were collected in the industrial area to evaluate its use as a bioindicator." + }, + { + "id": 14, + "value": "Plants of the genus Tillandsia were studied, and specimens of T. recurvata L. and T. tricholepis Baker were collected." + }, + { + "id": 15, + "value": "The specimens of T. recurvata L. were collected near the factory." } ] @@ -67,4 +103,52 @@ HTTP 200 "value": [ "Xanthoria parietina" ] -}] +}, +{ + "id": 7, + "value": [ + "Ramalina celastri", + "Tillandsia capillaris", + "Tillandsia recurvata", + "Tillandsia tricholepis" + ] +}, +{ + "id": 8, + "value": [] +}, +{ + "id": 9, + "value": [ + "Microgramma squamulosa" + ] +}, +{ + "id": 10, + "value": [] +}, +{ + "id": 11, + "value": [ + "Usnea hirta" + ] +}, +{ + "id": 12, + "value": [] +}, +{ + "id": 13, + "value": [] +}, +{ + "id": 14, + "value": [ + "Tillandsia recurvata", + "Tillandsia tricholepis" + ] +}, +{ + "id": 15, + "value": [] +}] \ No newline at end of file diff --git a/services/irc3-species/v1/IRC3sp.mjs b/services/irc3-species/v1/IRC3sp.mjs index db8f64fab..32a6328ce 100755 --- a/services/irc3-species/v1/IRC3sp.mjs +++ b/services/irc3-species/v1/IRC3sp.mjs @@ -100,7 +100,12 @@ const parseDocument = (line) => { // BINARY SEARCH // ============================================================================ -const findIndexInSortedArray = (key, arr, compareFn = (a, b) => a.localeCompare(b)) => { +// Binary search: Perl's `cmp` orders strings by code point, which is what +// JS string comparison operators do (locale collation would reorder accented +// keys and break the search). +const compareStrings = (a, b) => (a < b ? -1 : a > b ? 1 : 0); + +const findIndexInSortedArray = (key, arr, compareFn = compareStrings) => { let binf = -1; let bsup = arr.length; @@ -262,96 +267,68 @@ const createSpeciesExtractor = (table, pref, str, caseSensitive, logger) => { } /** - * Check if a matched form looks like a binomial name (contains a space) + * Canonical (table) form of a matched term. In the second pass, entries + * may be abbreviated forms, whose canonical and full forms come from the + * pass-specific maps instead of the resource table ones. */ - const isBinomial = (str) => str.includes(' '); + const canonicalOf = (term, maps) => (maps ? (maps.canonical[term] ?? str[term]) : str[term]); /** - * Find exact match in text + * Preferred form of a matched term. In the second pass, only abbreviated + * entries have one (their full form), like Perl's tmpPref hash. */ - const findExactMatch = (term, text) => { - const pattern = buildSearchPattern(term, caseSensitive); - const match = text.match(pattern); - if (!match || !isUppercase(match[0])) return null; - const found = match[0]; - return isBinomial(found) ? found : null; + const prefOf = (term, maps) => { + if (maps) return maps.abbreviation[term]; + return pref[term] ? str[pref[term]] : undefined; }; /** - * Get the canonical 2-word key (genus + species) from a table entry + * Build a result row: canonical form, found form, preferred form */ - const getTwoWordKey = (entry) => { - const words = entry.split(/\s+/); - return words.slice(0, 2).join(' '); + const buildMatchRow = (term, found, maps) => { + let row = `${canonicalOf(term, maps)}\t${found}`; + const preferred = prefOf(term, maps); + if (preferred) { + row += `\t${preferred}`; + } + return row; + }; + + /** + * Find exact match in text. Genus-only matches are kept: like in Perl, + * they are filtered out of the output, but they open the genus table for + * the second pass (abbreviation resolution). + */ + const findExactMatch = (term, text) => { + const pattern = buildSearchPattern(term, caseSensitive); + const match = text.match(pattern); + if (!match || !isUppercase(match[0])) return null; + return match[0]; }; /** - * Find partial matches for abbreviated genus - * Searches both forward and backward from startIndex + * Find the entry whose full normalized form prefixes the remaining text, + * walking the table backwards from the insertion point (Perl behaviour). + * Only whole terms are tested: matching a shorter form (e.g. the first two + * words of an infraspecific or virus name) must not yield the longer name. */ - const findPartialMatches = (text, searchTable, startIndex) => { + const findPartialMatches = (text, searchTable, startIndex, maps) => { const matches = []; if (!searchTable[startIndex]) return matches; const genusStart = splitOnWordBoundaries(searchTable[startIndex]).split(' ')[0]; - // Search backward from startIndex - const endIndex = Math.min(searchTable.length - 1, startIndex); - for (let i = endIndex; i >= 0; i--) { - const currentTerm = searchTable[i]; - const currentGenus = splitOnWordBoundaries(currentTerm).split(' ')[0]; - if (currentGenus !== genusStart) break; - - const testPatterns = []; - const twoWord = getTwoWordKey(currentTerm); - if (twoWord) testPatterns.push(twoWord); - testPatterns.push(currentTerm); - - for (const test of testPatterns) { - const escaped = escapeForRegex(test).replace(/\\ /g, '\\s*'); - const escapedPattern = escaped.replace(REGEX.NON_ASCII_CHARS, '.'); - const regex = new RegExp(`^${escapedPattern}\\b`, caseSensitive ? '' : 'i'); - const match = text.match(regex); - if (match && isUppercase(match[0])) { - const found = match[0]; - if (!isBinomial(found)) continue; - let result = `${str[currentTerm]}\t${found}`; - if (pref[currentTerm]) { - result += `\t${str[pref[currentTerm]]}`; - } - matches.push(result); - logger.debug(` -> Found: ${found}\n`); - return matches; - } - } - } - - // Search forward from startIndex - for (let i = startIndex; i < searchTable.length; i++) { + for (let i = startIndex; i >= 0; i--) { const currentTerm = searchTable[i]; const currentGenus = splitOnWordBoundaries(currentTerm).split(' ')[0]; if (currentGenus !== genusStart) break; - const testPatterns = []; - const twoWord = getTwoWordKey(currentTerm); - if (twoWord) testPatterns.push(twoWord); - testPatterns.push(currentTerm); - - for (const test of testPatterns) { - const escaped = escapeForRegex(test).replace(/\\ /g, '\\s*'); - const escapedPattern = escaped.replace(REGEX.NON_ASCII_CHARS, '.'); - const regex = new RegExp(`^${escapedPattern}\\b`, caseSensitive ? '' : 'i'); - const match = text.match(regex); - if (match && isUppercase(match[0])) { - const found = match[0]; - if (!isBinomial(found)) continue; - let result = `${str[currentTerm]}\t${found}`; - if (pref[currentTerm]) { - result += `\t${str[pref[currentTerm]]}`; - } - matches.push(result); - logger.debug(` -> Found: ${found}\n`); - return matches; - } + const regex = buildSearchPattern(currentTerm, caseSensitive); + const match = text.match(regex); + if (match && isUppercase(match[0])) { + const found = match[0]; + matches.push(buildMatchRow(currentTerm, found, maps)); + logger.debug(` -> Found: ${found}\n`); + return matches; } } @@ -360,8 +337,10 @@ const createSpeciesExtractor = (table, pref, str, caseSensitive, logger) => { /** * Find scientific names in text + * `searchTable` and `maps` allow a second pass on a reduced table with + * pass-specific canonical and preferred forms (abbreviation resolution). */ - const findScientificNames = (textToSearch, searchTable = table) => { + const findScientificNames = (textToSearch, searchTable = table, maps = null) => { let text = textToSearch.trim(); let rec = normalizeForLookup(text, caseSensitive); @@ -372,26 +351,19 @@ const createSpeciesExtractor = (table, pref, str, caseSensitive, logger) => { const normalizedRec = rec.trim(); if (!normalizedRec) break; - const recWords = normalizedRec.split(/\s+/); - const searchKey = recWords.slice(0, Math.min(2, recWords.length)).join(' '); - - const index = findIndexInSortedArray(searchKey, searchTable); + const index = findIndexInSortedArray(normalizedRec, searchTable); if (index > -1) { const term = searchTable[index]; const found = findExactMatch(term, text); if (found) { - let matchResult = `${str[term]}\t${found}`; - if (pref[term]) { - matchResult += `\t${str[pref[term]]}`; - } - matches.push(matchResult); + matches.push(buildMatchRow(term, found, maps)); } } else { const insertPos = -2 - index; if (insertPos >= 0 && insertPos < searchTable.length && searchTable[insertPos]) { - const partialMatches = findPartialMatches(text, searchTable, insertPos); + const partialMatches = findPartialMatches(text, searchTable, insertPos, maps); matches.push(...partialMatches); } } @@ -439,37 +411,49 @@ const createSpeciesExtractor = (table, pref, str, caseSensitive, logger) => { } } - const finalSearchTable = uniqueAndSort(expandedTerms); - - /** @type {Record} */ - const abbreviationMap = {}; + // Second-pass table: every candidate term plus its abbreviated form, + // so that abbreviated occurrences (e.g. “T. recurvata”) can be found, + // with pass-specific canonical and full forms (Perl's tmpStr/tmpPref). + const finalSearchTerms = []; /** @type {Record} */ const canonicalMap = {}; + /** @type {Record} */ + const abbreviationMap = {}; - for (const term of finalSearchTable) { + for (const term of uniqueAndSort(expandedTerms)) { if (!term || !str[term]) continue; + finalSearchTerms.push(term); canonicalMap[normalizeForLookup(term, caseSensitive)] = str[term]; const abbrev = buildAbbreviation(term); if (abbrev) { const abbrevKey = normalizeForLookup(abbrev, caseSensitive); - const abbrevNormalized = caseSensitive ? abbrev : abbrev.toLowerCase(); - const canonicalForm = caseSensitive ? abbrev : abbrev.charAt(0).toUpperCase() + abbrev.slice(1); - abbreviationMap[abbrevKey] = str[term]; + finalSearchTerms.push(abbrevKey); canonicalMap[abbrevKey] = canonicalForm; + + if (abbreviationMap[abbrevKey]) { + abbreviationMap[abbrevKey] += ` ; ${str[term]}`; + } else { + abbreviationMap[abbrevKey] = str[term]; + } } } + const finalSearchTable = uniqueAndSort(finalSearchTerms); + /** @type {string[]} */ const resolvedMatches = []; for (const para of refPara) { - const found = findScientificNames(para, finalSearchTable); + const found = findScientificNames(para, finalSearchTable, { + canonical: canonicalMap, + abbreviation: abbreviationMap, + }); for (const match of found) { if (!match) continue; @@ -505,14 +489,14 @@ const createSpeciesExtractor = (table, pref, str, caseSensitive, logger) => { if (seen[canonical]) continue; seen[canonical] = true; - const [, foundForm, prefForm] = result.split('\t'); - const formatted = `${foundForm}\t${canonical}\t${pref[canonical] || ''}`; + const [, foundForm, fullForm] = result.split('\t'); + const formatted = `${foundForm ?? ''}\t${canonical}\t${fullForm ?? ''}`; logger.debug(`\r`); output.push(formatted); - if (prefForm && prefForm.match(/^\?.+\?$/) && logger) { + if (fullForm && fullForm.match(/^\?.+\?$/) && logger) { const msg = `WARNING! ${id}: ambiguity on non-abbreviated form of "${canonical}"!\n`; logger.error(msg); logger.writeLog(msg); @@ -548,6 +532,8 @@ const processDocument = (doc, extractor) => { return canonical; }) .filter(Boolean) + // Ambiguous abbreviated forms must not yield a species (Perl's passe1) + .filter(name => !/^\?.+\?$/.test(name)) .filter(name => name.includes(' ')) .filter((name, index, arr) => arr.indexOf(name) === index) .sort(); @@ -588,9 +574,30 @@ const processJsonlStream = async (extractor) => { // MAIN // ============================================================================ +/** + * Command line options, mirroring the Perl script ones: -t table, -c casse. + * Other options (-w, -j, -f, ...) are accepted for compatibility but unused: + * the script always reads JSON lines from stdin. + */ +const parseArgs = (argv) => { + const options = { casse: false, table: undefined }; + + for (let i = 0; i < argv.length; i++) { + const arg = argv[i]; + if (arg === '-t' || arg === '--table') { + options.table = argv[++i]; + } else if (arg === '-c' || arg === '--casse') { + options.casse = true; + } + } + + return options; +}; + const main = async () => { - const tablePath = process.env.IRC3SP_TABLE || DEFAULT_TABLE_PATH; - const caseSensitive = process.env.IRC3SP_CASE_SENSITIVE === 'true'; + const options = parseArgs(process.argv.slice(2)); + const tablePath = options.table ?? process.env.IRC3SP_TABLE ?? DEFAULT_TABLE_PATH; + const caseSensitive = options.casse || process.env.IRC3SP_CASE_SENSITIVE === 'true'; const logger = createLogger(process.env); if (!tablePath) { diff --git a/services/irc3-species/v1/irc3sp.ini b/services/irc3-species/v1/irc3sp.ini index 6732a414a..6e6a25715 100644 --- a/services/irc3-species/v1/irc3sp.ini +++ b/services/irc3-species/v1/irc3sp.ini @@ -54,7 +54,7 @@ plugin = @ezs/strings [JSONParse] -[sentences] +[STRSentences] path = env('path', 'value') [expand]