From 8aeb1150d04907ec1f04959fa2d80c3afad71cc5 Mon Sep 17 00:00:00 2001 From: leogail Date: Fri, 13 Mar 2026 13:24:18 +0100 Subject: [PATCH 1/9] feat(text-lucene): create service --- README.md | 2 ++ package.json | 1 + services/text-lucene/.dockerignore | 7 ++++++ services/text-lucene/Dockerfile | 18 +++++++++++++++ services/text-lucene/README.md | 5 ++++ services/text-lucene/config.json | 14 +++++++++++ services/text-lucene/examples.http | 17 ++++++++++++++ services/text-lucene/package.json | 37 ++++++++++++++++++++++++++++++ services/text-lucene/swagger.json | 33 ++++++++++++++++++++++++++ 9 files changed, 134 insertions(+) create mode 100644 services/text-lucene/.dockerignore create mode 100644 services/text-lucene/Dockerfile create mode 100644 services/text-lucene/README.md create mode 100644 services/text-lucene/config.json create mode 100644 services/text-lucene/examples.http create mode 100644 services/text-lucene/package.json create mode 100644 services/text-lucene/swagger.json diff --git a/README.md b/README.md index b9a867103..5653bc5a2 100644 --- a/README.md +++ b/README.md @@ -35,6 +35,7 @@ All contributing instructions are in [CONTRIBUTING](CONTRIBUTING.md). - [data-computer](./services/data-computer) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-data-computer.svg)](https://hub.docker.com/r/cnrsinist/ws-data-computer/) - [data-graph](./services/data-graph) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-data-graph.svg)](https://hub.docker.com/r/cnrsinist/ws-data-graph/) - [data-homogenise](./services/data-homogenise) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-data-homogenise.svg)](https://hub.docker.com/r/cnrsinist/ws-data-homogenise/) +- [data-kwsimilarity](./services/data-kwsimilarity) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-data-kwsimilarity.svg)](https://hub.docker.com/r/cnrsinist/ws-data-kwsimilarity/) - [data-rapido](./services/data-rapido) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-data-rapido.svg)](https://hub.docker.com/r/cnrsinist/ws-data-rapido/) - [data-table](./services/data-table) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-data-table.svg)](https://hub.docker.com/r/cnrsinist/ws-data-table/) - [data-termsuite](./services/data-termsuite) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-data-termsuite.svg)](https://hub.docker.com/r/cnrsinist/ws-data-termsuite/) @@ -59,5 +60,6 @@ All contributing instructions are in [CONTRIBUTING](CONTRIBUTING.md). - [terms-extraction](./services/terms-extraction) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-terms-extraction.svg)](https://hub.docker.com/r/cnrsinist/ws-terms-extraction/) - [terms-tools](./services/terms-tools) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-terms-tools.svg)](https://hub.docker.com/r/cnrsinist/ws-terms-tools/) - [text-clustering](./services/text-clustering) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-text-clustering.svg)](https://hub.docker.com/r/cnrsinist/ws-text-clustering/) +- [text-lucene](./services/text-lucene) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-text-lucene.svg)](https://hub.docker.com/r/cnrsinist/ws-text-lucene/) - [text-summarize](./services/text-summarize) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-text-summarize.svg)](https://hub.docker.com/r/cnrsinist/ws-text-summarize/) - [transli-tal](./services/transli-tal) [![Docker Pulls](https://img.shields.io/docker/pulls/cnrsinist/ws-transli-tal.svg)](https://hub.docker.com/r/cnrsinist/ws-transli-tal/) diff --git a/package.json b/package.json index 0058ff2d0..1ee4b5f16 100644 --- a/package.json +++ b/package.json @@ -78,6 +78,7 @@ "services/terms-extraction", "services/terms-tools", "services/text-clustering", + "services/text-lucene", "services/text-summarize", "services/transli-tal" ], diff --git a/services/text-lucene/.dockerignore b/services/text-lucene/.dockerignore new file mode 100644 index 000000000..e280cbb78 --- /dev/null +++ b/services/text-lucene/.dockerignore @@ -0,0 +1,7 @@ +# Ignore all files by default +* + +# White list only the required files +!config.json +!v1 +!swagger.json diff --git a/services/text-lucene/Dockerfile b/services/text-lucene/Dockerfile new file mode 100644 index 000000000..c4d7bd539 --- /dev/null +++ b/services/text-lucene/Dockerfile @@ -0,0 +1,18 @@ +# syntax=docker/dockerfile:1.2 +FROM cnrsinist/ezs-python-server:py3.9-no24-1.0.13 + +WORKDIR /app +# Install all python dependencies +# RUN pip install --no-cache-dir \ +# unidecode==1.3.7 + +# Install all node dependencies +# RUN npm install --omit=dev \ +# @ezs/teeft@2.3.2 \ +# @ezs/strings@1.0.5 && \ +# npm cache clean --force + +WORKDIR /app/public +# Declare files to copy in .dockerignore +COPY --chown=daemon:daemon . /app/public/ +COPY --chown=daemon:daemon ./config.json /app/config.json diff --git a/services/text-lucene/README.md b/services/text-lucene/README.md new file mode 100644 index 000000000..e77057db2 --- /dev/null +++ b/services/text-lucene/README.md @@ -0,0 +1,5 @@ +# ws-text-lucene@0.0.0 + +Assistance à la génération de requête dans Istex Search + +Génère une requête en syntaxe Lucene a partir d'une requête en langue naturelle (multilingue) diff --git a/services/text-lucene/config.json b/services/text-lucene/config.json new file mode 100644 index 000000000..c0cb140c7 --- /dev/null +++ b/services/text-lucene/config.json @@ -0,0 +1,14 @@ +{ + "environnement": { + "EZS_TITLE": "Assistance à la génération de requête dans Istex Search", + "EZS_DESCRIPTION": "Génère une requête en syntaxe Lucene a partir d'une requête en langue naturelle (multilingue)", + "EZS_METRICS": true, + "EZS_CONCURRENCY": 2, + "EZS_CONTINUE_DELAY": 60, + "EZS_NSHARDS": 32, + "EZS_CACHE": true, + "EZS_VERBOSE": false, + "NODE_OPTIONS": "--max_old_space_size=1024", + "NODE_ENV": "production" + } +} \ No newline at end of file diff --git a/services/text-lucene/examples.http b/services/text-lucene/examples.http new file mode 100644 index 000000000..837d964ff --- /dev/null +++ b/services/text-lucene/examples.http @@ -0,0 +1,17 @@ +# These examples can be used directly in VSCode, using HTTPYac extension (anweber.vscode-httpyac) +# They are important, because used to generate the tests.hurl file. + +# Décommenter/commenter les lignes voulues pour tester localement +@host=http://localhost:31976 +# @host=https://text-lucene.services.istex.fr + +### +# @name v1routeInCamelCase +# Description de la route +POST {{host}}/v1/route/in/camel/case?indent=true HTTP/1.1 +Content-Type: application/json + +[ + { "value": "une valeur typique" }, + { "value": "en json" } +] diff --git a/services/text-lucene/package.json b/services/text-lucene/package.json new file mode 100644 index 000000000..7357c5791 --- /dev/null +++ b/services/text-lucene/package.json @@ -0,0 +1,37 @@ +{ + "private": true, + "name": "ws-text-lucene", + "version": "0.0.0", + "description": "Assistance à la génération de requête dans Istex Search", + "repository": { + "type": "git", + "url": "git+https://github.com/Inist-CNRS/web-services.git" + }, + "keywords": [ + "ezmaster" + ], + "author": "Léo Gaillard ", + "license": "MIT", + "bugs": { + "url": "https://github.com/Inist-CNRS/web-services/issues" + }, + "homepage": "https://github.com/Inist-CNRS/web-services/#readme", + "scripts": { + "version:insert:readme": "sed -i \"s#\\(${npm_package_name}.\\)\\([\\.a-z0-9]\\+\\)#\\1${npm_package_version}#g\" README.md && git add README.md", + "version:insert:swagger": "sed -i \"s/\\\"version\\\": \\\"[0-9]\\+.[0-9]\\+.[0-9]\\+\\\"/\\\"version\\\": \\\"${npm_package_version}\\\"/g\" swagger.json && git add swagger.json", + "version:insert": "npm run version:insert:readme && npm run version:insert:swagger", + "version:commit": "git commit -a -m \"release ${npm_package_name}@${npm_package_version}\"", + "version:tag": "git tag \"${npm_package_name}@${npm_package_version}\" -m \"${npm_package_name}@${npm_package_version}\"", + "version:push": "git push && git push --tags", + "version": "npm run version:insert && npm run version:commit && npm run version:tag", + "postversion": "npm run version:push", + "build:check": "DOCKER_BUILDKIT=1 docker build --check -t cnrsinist/${npm_package_name}:latest . && docker run --rm -i hadolint/hadolint hadolint - < Dockerfile ", + "build:dev": "docker build -t cnrsinist/${npm_package_name}:latest .", + "start:dev": "npm run build:dev && docker run --name dev --rm --detach -p 31976:31976 cnrsinist/${npm_package_name}:latest", + "stop:dev": "docker stop dev", + "build": "docker build -t cnrsinist/${npm_package_name}:${npm_package_version} .", + "start": "docker run --rm -p 31976:31976 cnrsinist/${npm_package_name}:${npm_package_version}", + "publish": "docker push cnrsinist/${npm_package_name}:${npm_package_version}" + }, + "avoid-testing": false +} \ No newline at end of file diff --git a/services/text-lucene/swagger.json b/services/text-lucene/swagger.json new file mode 100644 index 000000000..0e8804b05 --- /dev/null +++ b/services/text-lucene/swagger.json @@ -0,0 +1,33 @@ +{ + "openapi": "3.0.0", + "info": { + "title": "text-lucene - Assistance à la génération de requête dans Istex Search", + "description": "Génère une requête en syntaxe Lucene a partir d'une requête en langue naturelle (multilingue)", + "version": "0.0.0", + "termsOfService": "https://services.istex.fr/", + "contact": { + "name": "Inist-CNRS", + "url": "https://www.inist.fr/nous-contacter/" + } + }, + "servers": [ + { + "x-comment": "Will be automatically completed by the ezs server." + }, + { + "url": "http://vptdmservices.intra.inist.fr:49225/", + "description": "Latest version for production", + "#DISABLED#x-profil": "Standard" + } + ], + "tags": [ + { + "name": "text-lucene", + "description": "Assistance à la génération de requête dans Istex Search", + "externalDocs": { + "description": "Plus de documentation", + "url": "https://github.com/inist-cnrs/web-services/tree/main/services/text-lucene" + } + } + ] +} \ No newline at end of file From ee2710e8225d8156ff4a2e5d20fb009d88241777 Mon Sep 17 00:00:00 2001 From: leogail Date: Fri, 13 Mar 2026 13:25:21 +0100 Subject: [PATCH 2/9] feat(text-lucene): Use ilaas git secret as venv --- .github/workflows/test-and-publish-on-tag.yml | 4 ++-- .github/workflows/test-on-branch.yml | 2 +- bin/create-env.sh | 2 ++ 3 files changed, 5 insertions(+), 3 deletions(-) diff --git a/.github/workflows/test-and-publish-on-tag.yml b/.github/workflows/test-and-publish-on-tag.yml index fde4c3f00..7423b168c 100644 --- a/.github/workflows/test-and-publish-on-tag.yml +++ b/.github/workflows/test-and-publish-on-tag.yml @@ -28,7 +28,7 @@ jobs: - name: Build .env from Github Actions secrets shell: bash run: | - bin/create-env.sh ${{ github.ref_name }} ${{ secrets.WEBDAV_LOGIN }} ${{ secrets.WEBDAV_PASSWORD }} ${{ secrets.WEBDAV_URL }} ${{secrets.OPENALEX_API_KEY}} ${{secrets.UNPAYWALL_API_KEY}} ${{secrets.CROSSREF_API_KEY}} ${{secrets.LIBRARIES_IO_API_KEY}} ${{secrets.ORCID_CLIENT_ID}} ${{secrets.ORCID_SECRET}} + bin/create-env.sh ${{ github.ref_name }} ${{ secrets.WEBDAV_LOGIN }} ${{ secrets.WEBDAV_PASSWORD }} ${{ secrets.WEBDAV_URL }} ${{secrets.OPENALEX_API_KEY}} ${{secrets.UNPAYWALL_API_KEY}} ${{secrets.CROSSREF_API_KEY}} ${{secrets.LIBRARIES_IO_API_KEY}} ${{secrets.ORCID_CLIENT_ID}} ${{secrets.ORCID_SECRET}} ${{secrets.ILAAS_API_KEY}} ls -l services/*/.env - name: Build Docker Image & Test @@ -58,7 +58,7 @@ jobs: - name: Build .env from Github Actions secrets shell: bash run: | - bin/create-env.sh ${{ github.ref_name }} ${{ secrets.WEBDAV_LOGIN }} ${{ secrets.WEBDAV_PASSWORD }} ${{ secrets.WEBDAV_URL }} ${{secrets.OPENALEX_API_KEY}} ${{secrets.UNPAYWALL_API_KEY}} ${{secrets.CROSSREF_API_KEY}} ${{secrets.LIBRARIES_IO_API_KEY}} ${{secrets.ORCID_CLIENT_ID}} ${{secrets.ORCID_SECRET}} + bin/create-env.sh ${{ github.ref_name }} ${{ secrets.WEBDAV_LOGIN }} ${{ secrets.WEBDAV_PASSWORD }} ${{ secrets.WEBDAV_URL }} ${{secrets.OPENALEX_API_KEY}} ${{secrets.UNPAYWALL_API_KEY}} ${{secrets.CROSSREF_API_KEY}} ${{secrets.LIBRARIES_IO_API_KEY}} ${{secrets.ORCID_CLIENT_ID}} ${{secrets.ORCID_SECRET}} ${{secrets.ILAAS_API_KEY}} ls -l services/*/.env - name: Build & Push Docker Image diff --git a/.github/workflows/test-on-branch.yml b/.github/workflows/test-on-branch.yml index 276e9fd2b..c6b40e2bb 100644 --- a/.github/workflows/test-on-branch.yml +++ b/.github/workflows/test-on-branch.yml @@ -32,7 +32,7 @@ jobs: - name: Build .env from Github Actions secrets shell: bash run: | - bin/create-env.sh ${{ github.ref_name }} ${{ secrets.WEBDAV_LOGIN }} ${{ secrets.WEBDAV_PASSWORD }} ${{ secrets.WEBDAV_URL }} ${{secrets.OPENALEX_API_KEY}} ${{secrets.UNPAYWALL_API_KEY}} ${{secrets.CROSSREF_API_KEY}} ${{secrets.LIBRARIES_IO_API_KEY}} ${{secrets.ORCID_CLIENT_ID}} ${{secrets.ORCID_SECRET}} + bin/create-env.sh ${{ github.ref_name }} ${{ secrets.WEBDAV_LOGIN }} ${{ secrets.WEBDAV_PASSWORD }} ${{ secrets.WEBDAV_URL }} ${{secrets.OPENALEX_API_KEY}} ${{secrets.UNPAYWALL_API_KEY}} ${{secrets.CROSSREF_API_KEY}} ${{secrets.LIBRARIES_IO_API_KEY}} ${{secrets.ORCID_CLIENT_ID}} ${{secrets.ORCID_SECRET}} ${{secrets.ILAAS_API_KEY}} ls -l services/*/.env - name: Build Docker Image & Test diff --git a/bin/create-env.sh b/bin/create-env.sh index bbd37c57c..dfe06fbb2 100755 --- a/bin/create-env.sh +++ b/bin/create-env.sh @@ -32,6 +32,7 @@ CROSSREF_API_KEY=$7 LIBRARIES_IO_API_KEY=$8 ORCID_CLIENT_ID=$9 ORCID_SECRET=${10} +ILAAS_API_KEY=${11} echo "Building .env for $SERVICE_NAME" @@ -45,4 +46,5 @@ echo "Building .env for $SERVICE_NAME" echo "export LIBRARIES_IO_API_KEY=$LIBRARIES_IO_API_KEY" echo "export ORCID_CLIENT_ID=$ORCID_CLIENT_ID" echo "export ORCID_SECRET=$ORCID_SECRET" + echo "export ILAAS_API_KEY=$ILAAS_API_KEY" } > "services/$SERVICE_NAME/.env" From 6e2f5274f960b2cd1180bc61a8b5726c28871676 Mon Sep 17 00:00:00 2001 From: leogail Date: Fri, 13 Mar 2026 16:37:08 +0100 Subject: [PATCH 3/9] feat(text-lucene): add base code --- services/text-lucene/Dockerfile | 10 +- services/text-lucene/config.json | 5 +- services/text-lucene/examples.http | 14 +- services/text-lucene/package.json | 4 +- services/text-lucene/v1/istex-search.ini | 36 +++++ services/text-lucene/v1/istex-search.py | 178 +++++++++++++++++++++++ 6 files changed, 230 insertions(+), 17 deletions(-) create mode 100644 services/text-lucene/v1/istex-search.ini create mode 100755 services/text-lucene/v1/istex-search.py diff --git a/services/text-lucene/Dockerfile b/services/text-lucene/Dockerfile index c4d7bd539..7e8c8a5ca 100644 --- a/services/text-lucene/Dockerfile +++ b/services/text-lucene/Dockerfile @@ -3,14 +3,8 @@ FROM cnrsinist/ezs-python-server:py3.9-no24-1.0.13 WORKDIR /app # Install all python dependencies -# RUN pip install --no-cache-dir \ -# unidecode==1.3.7 - -# Install all node dependencies -# RUN npm install --omit=dev \ -# @ezs/teeft@2.3.2 \ -# @ezs/strings@1.0.5 && \ -# npm cache clean --force +RUN pip install --no-cache-dir \ + requests==2.32.5 WORKDIR /app/public # Declare files to copy in .dockerignore diff --git a/services/text-lucene/config.json b/services/text-lucene/config.json index c0cb140c7..570db747e 100644 --- a/services/text-lucene/config.json +++ b/services/text-lucene/config.json @@ -9,6 +9,7 @@ "EZS_CACHE": true, "EZS_VERBOSE": false, "NODE_OPTIONS": "--max_old_space_size=1024", - "NODE_ENV": "production" + "NODE_ENV": "production", + "ILAAS_API_KEY": "real_api_key" } -} \ No newline at end of file +} diff --git a/services/text-lucene/examples.http b/services/text-lucene/examples.http index 837d964ff..236c9c0b7 100644 --- a/services/text-lucene/examples.http +++ b/services/text-lucene/examples.http @@ -6,12 +6,16 @@ # @host=https://text-lucene.services.istex.fr ### -# @name v1routeInCamelCase -# Description de la route -POST {{host}}/v1/route/in/camel/case?indent=true HTTP/1.1 +# @name v1IstexSearch +# Construit une requête syntaxe Lucene à partir d'une requête en langue naturelle (quelle que soit la langue). +POST {{host}}/v1/istex-search?indent=true HTTP/1.1 Content-Type: application/json [ - { "value": "une valeur typique" }, - { "value": "en json" } + { + "value": "Je veux un corpus d'articles scientifiques publiés post 2020 sur la biodiversité dans la méditerranée nord. Les documents seront en anglais" + }, + { + "value": "no query" + } ] diff --git a/services/text-lucene/package.json b/services/text-lucene/package.json index 7357c5791..dbce2c1f5 100644 --- a/services/text-lucene/package.json +++ b/services/text-lucene/package.json @@ -27,11 +27,11 @@ "postversion": "npm run version:push", "build:check": "DOCKER_BUILDKIT=1 docker build --check -t cnrsinist/${npm_package_name}:latest . && docker run --rm -i hadolint/hadolint hadolint - < Dockerfile ", "build:dev": "docker build -t cnrsinist/${npm_package_name}:latest .", - "start:dev": "npm run build:dev && docker run --name dev --rm --detach -p 31976:31976 cnrsinist/${npm_package_name}:latest", + "start:dev": ". ./.env 2> /dev/null; npm run build:dev && docker run -e ILAAS_API_KEY --name dev --rm --detach -p 31976:31976 cnrsinist/${npm_package_name}:latest", "stop:dev": "docker stop dev", "build": "docker build -t cnrsinist/${npm_package_name}:${npm_package_version} .", "start": "docker run --rm -p 31976:31976 cnrsinist/${npm_package_name}:${npm_package_version}", "publish": "docker push cnrsinist/${npm_package_name}:${npm_package_version}" }, "avoid-testing": false -} \ No newline at end of file +} diff --git a/services/text-lucene/v1/istex-search.ini b/services/text-lucene/v1/istex-search.ini new file mode 100644 index 000000000..a40b47ce4 --- /dev/null +++ b/services/text-lucene/v1/istex-search.ini @@ -0,0 +1,36 @@ +# OpenAPI Documentation - JSON format (dot notation) +mimeType = application/json + +post.operationId = post-v1-istex-search +post.summary = Assistance à la génération de requête Lucene dans Istex Search +post.description = Génère une requête en syntaxe Lucene a partir d'une requête en langue naturelle (service multilingue) +post.tags.0 = text-lucene +post.requestBody.content.application/json.schema.$ref = #/components/schemas/JSONStream +post.requestBody.required = true +post.responses.default.content.application/json.schema.$ref = #/components/schemas/JSONStream +post.responses.default.description = Le champ `value` contient la requête Lucene générée +post.parameters.0.description = Indenter le JSON résultant +post.parameters.0.in = query +post.parameters.0.name = indent +post.parameters.0.schema.type = boolean +#' + +# Examples + +[use] +plugin = @ezs/spawn +plugin = @ezs/basics + +[JSONParse] +separator = * + +[expand] +path = value +size = 1 + +[expand/exec] +# command should be executable ! +command = ./v1/istex-search.py + +[dump] +indent = env('indent', false) diff --git a/services/text-lucene/v1/istex-search.py b/services/text-lucene/v1/istex-search.py new file mode 100755 index 000000000..e657d60c9 --- /dev/null +++ b/services/text-lucene/v1/istex-search.py @@ -0,0 +1,178 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- + +import requests +import json +import sys +import os +import time + +api_key = os.getenv("ILAAS_API_KEY") +model = "gpt-oss-120b" + + +def construct_llm_prompt(user_prompt): + # Documentation des champs Istex + fields_doc = """ + - `abstract:()` pour rechercher des mots-clés dans l'abstract + - `title:()` pour rechercher des mots-clés dans le titre + - `subject.value:()` pour rechercher des mots-clés parmis ceux renseignés par les auteurs + - `publicationDate:[N TO M]` : pour rechercher entre l'année N et M. + - `publicationDate:[N TO *]` : pour obtenir les documents publiés uniquement après l'année N. + - `author.name.raw:()` : pour inclure des noms d'auteurs spécifiquement. Uniquement le **nom de famille**. + - `language.raw:()` : permet de choisir la langues de documents ("fre" pour français, "eng" pour anglais, "deu" pour allemand et "spa" pour espagnol.) + - `AND genre.raw:"research-article"` permet d'avoir uniquement les articles de recherche si l'utilisateur veut quelque chose de filtré. + """ + + prompt = f""" + Tu es un assistant spécialisé dans la génération de requêtes en syntaxe Lucene. Voici les règles strictes à suivre : + + ### Contexte : + **Prompt utilisateur** : {user_prompt} + **Liste des champs disponibles** : {fields_doc} + + ### Consignes : + 1. **Analyse du prompt utilisateur** : + - Identifie les mots-clés, les champs mentionnés explicitement, et les intentions de recherche. + - Pour la langue, on récupère la ou les langue(s) demandée(s) par l'utilisateur. Si aucune langue n'est demandé, n'utilise **pas** le champ `language.raw`. + - Si aucun mot-clé n'est présent, extrapole des mots-clés scientifiques pertinents à partir du contexte, et leur variantes (pluriels, féminins) + - Les mots clés seront présents dans les langues demandées par l'utilisateur. Si aucune langue n'est demandée, génères des mots clés dans la langue de la requête **et** en anglais. + - une fois les mots-clés déterminés, ils seront systématiquement recherchés dans le titre, dans l'abstract et dans `subject.value` sauf mention contraire de l'utilisateur. + - Répond très précisément à la demande de l'utilisateur. + + 2. **Respect des champs** : Ne **jamais** inventer ou ajouter des champs non documentés. + + 3. **Génération de la requête Lucene** : Construis une requête syntaxiquement correcte en Lucene, en utilisant les champs et mots-clés identifiés. Utilise les opérateurs Lucene appropriés (AND, OR, NOT, etc.) pour refléter la logique de la demande utilisateur. + + 4. **Contrôle de la sortie** : Ne retourne que la requête Lucene finale **encadrée uniquement de triple quote**, sans explication ni commentaire. La sortie doit être **strictement** au format : + ``` + [Requête Lucene valide] + ``` + + ### Exemple de sortie attendue : + 1. Exemple 1 : Si l'utilisateur demande "I want to create a recent corpus (publications released after 2015) on electric cars", + une réponse possible est + ``` + (title:("electric car" "electrics cars" "electric vehicle" "electrics vehicles") OR abstract:("electric car" "electrics cars" "electric vehicle" "electrics vehicles")) AND publicationDate:[2015 TO *] + ``` + + 2. Exemple 2 : Si l'utilisateur demande "Trouve tous les articles scientifiques parûts entre 2000 et 2020 sur la mémoire", + une réponse possible est + ``` + (title:("memory" "memories" "metamemory" "mémoire" "mémoires" "métamémoire") OR abstract:("memory" "memories" "metamemory" "mémoire" "mémoires" "métamémoire")) AND publicationDate:[2000 TO 2020] AND genre.raw:("research-article") + ``` + + 3. Exemple 3 : Si l'utilisateur demande "Je veux créer un corpus à partir de ces mots-clés : réchauffement climatique, réchauffement planétaire, réchauffement global, changement climatique, dérèglement climatique. Les documents peuvent être en anglais ou en français." + Ici les langues sont spécifiées. Après traduction des mots clés données dans les langues souhaitées, une réponse possible est + ``` + (title:("réchauffement climatique" "réchauffement planétaire" "réchauffement global" "changement climatique" "dérèglement climatique" "global warming" "climate warming" "climate disruption") OR abstract:("réchauffement climatique" "réchauffement planétaire" "réchauffement global" "changement climatique" "dérèglement climatique" "global warming" "climate warming" "climate disruption")) AND language.raw:("eng" "fre") + ``` + + 4. Exemple 4 : Si l'utilisateur demande "réchauffement climatique, réchauffement planétaire, réchauffement global, changement climatique, dérèglement climatique." + Bien que la langue ne soit pas spécifiée, l'utilisateur donne juste une liste de mots-clés précis dans une seule langue précise : on utilisera seulement celle-ci. Une réponse possible est + ``` + (title:("réchauffement climatique" "réchauffement planétaire" "réchauffement global" "changement climatique" "dérèglement climatique") OR abstract:("réchauffement climatique" "réchauffement planétaire" "réchauffement global" "changement climatique" "dérèglement climatique")) AND language.raw:("fre") + ``` + + 5. Exemple 5 : Si l'utilisateur demande "Je souhaite récupérer les études de Léon Gaillad ou J. Rebol sur la traduction automatique." + En ne récupérant que les noms des auteurs, une réponse possible est : + ``` + (title:("machine translation" "automatic translation" "automated translation" "traduction automatique") OR abstract:("machine translation" "automatic translation" "automated translation" "traduction automatique")) AND author.name.raw:("Gaillad" "Rebol") + ``` + + 6. Exemple 6 : si l'utilisateur demande "What documents discuss the impact of screen time on the mental health ? Spanish or english documents only." + Après voir séparé les deux notions qui doivent apparaître toutes deux (**mental health** et **screen time**), une réponse possible est + ``` + (title:("screen time" "time on screen" "tiempo de pantalla" "tiempo frente a la pantalla" "uso de pantallas") OR abstract:("screen time" "time on screen" "tiempo de pantalla" "tiempo frente a la pantalla" "uso de pantallas")) AND (title:("mental health" "mental disorder" "mental disorders" "salud mental" "trastorno mental" "trastornos mentales" "enfermedad mental" "enfermedades mentales") OR abstract:("mental health" "mental disorder" "mental disorders" "salud mental" "trastorno mental" "trastornos mentales" "enfermedad mental" "enfermedades mentales")) AND language.raw:("eng" "spa") + ``` + + 7. Exemple 7 : si l'utilisateur demande "Je veux un corpus d'articles scientifiques publiés post 2020 sur la biodiversité dans la méditerranée nord. Les documents seront en anglais." + Après voir séparé les deux notions qui doivent apparaître toutes deux (la **biodiversité** et **la méditerranée nord**), une réponse possible est + ``` + (title:("biodiversity" "marine biodiversity" "ocean biodiversity" "oceanic biodiversity") OR abstract:("biodiversity" "marine biodiversity" "ocean biodiversity" "oceanic biodiversity")) AND (title:("north mediterranean" "north mediterranean sea") OR abstract:("north mediterranean" "north mediterranean sea")) AND language.raw:("eng") AND genre.raw:"research-article" AND publicationDate:[2020 TO *] + ``` + + ### Exécution : + Génère maintenant la requête Lucene pour le prompt utilisateur fourni : + """ + + return prompt + + +def generate_lucen(prompt: str, model_name: str) -> str: + base_url = "https://llm.ilaas.fr/v1" + headers = { + "Authorization": f"Bearer {api_key}", + "Content-Type": "application/json" + } + payload = { + "model": model_name, + "messages": [{"role": "user", "content": f"{prompt}"}], + "stream": False, + "max_tokens": 2000 + } + response = requests.post( + f"{base_url}/chat/completions", + headers=headers, + json=payload) + + result = response.json() + + return result['choices'][0]['message']['content'] + + +def process_llm_response(llm_output): + lucene_equation = llm_output.split("```")[1] + return lucene_equation.replace("\n", " ").replace(" ", " ").strip() + + +class NoAnswerError(Exception): + pass + + +def main(): + retries = 3 + for line in sys.stdin: + try: + user_prompt = json.loads(line) + value = user_prompt["value"] + if len(value) < 10: + raise Exception("To short request") + prompt = construct_llm_prompt(value) + output = None + + for retry in range(retries): + try: + response = generate_lucen(prompt, model_name=model) + response = process_llm_response(response) + # If model doesn't generate "```", + # this function returns an IdexError. + if len(response) < 10: + # We consider the response to short to be a lucene eq. + raise NoAnswerError("Empty output") + output = response + break + except (NoAnswerError, IndexError): + if retry < retries - 1: + time.sleep(2 * (retry + 1)) + continue + except Exception: + output = None + break + + if output is None: + output = "" + + user_prompt["value"] = output + + except Exception as e: + sys.stderr.write(f"Unexpected error: {str(e)}") + sys.stderr.write("\n") + user_prompt = {"value": ""} + + sys.stdout.write(json.dumps(user_prompt)) + sys.stdout.write("\n") + + +if __name__ == "__main__": + main() From 7135696847c993869cc9180718b041996da6ac65 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?L=C3=A9o=20Gaillard?= <115534900+leogail@users.noreply.github.com> Date: Mon, 16 Mar 2026 08:14:39 +0100 Subject: [PATCH 4/9] Apply suggestions from code review MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-authored-by: François Parmentier --- services/text-lucene/README.md | 2 +- services/text-lucene/config.json | 2 +- services/text-lucene/swagger.json | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/services/text-lucene/README.md b/services/text-lucene/README.md index e77057db2..e35aecbc4 100644 --- a/services/text-lucene/README.md +++ b/services/text-lucene/README.md @@ -2,4 +2,4 @@ Assistance à la génération de requête dans Istex Search -Génère une requête en syntaxe Lucene a partir d'une requête en langue naturelle (multilingue) +Génère une requête en syntaxe Lucene à partir d'une requête en langue naturelle (multilingue). C'est la syntaxe utilisée par Istex Search (et par l'API Istex). diff --git a/services/text-lucene/config.json b/services/text-lucene/config.json index 570db747e..a2e14216a 100644 --- a/services/text-lucene/config.json +++ b/services/text-lucene/config.json @@ -1,7 +1,7 @@ { "environnement": { "EZS_TITLE": "Assistance à la génération de requête dans Istex Search", - "EZS_DESCRIPTION": "Génère une requête en syntaxe Lucene a partir d'une requête en langue naturelle (multilingue)", + "EZS_DESCRIPTION": "Génère une requête en syntaxe Lucene à partir d'une requête en langue naturelle (multilingue)", "EZS_METRICS": true, "EZS_CONCURRENCY": 2, "EZS_CONTINUE_DELAY": 60, diff --git a/services/text-lucene/swagger.json b/services/text-lucene/swagger.json index 0e8804b05..50f0120a0 100644 --- a/services/text-lucene/swagger.json +++ b/services/text-lucene/swagger.json @@ -2,7 +2,7 @@ "openapi": "3.0.0", "info": { "title": "text-lucene - Assistance à la génération de requête dans Istex Search", - "description": "Génère une requête en syntaxe Lucene a partir d'une requête en langue naturelle (multilingue)", + "description": "Génère une requête en syntaxe Lucene à partir d'une requête en langue naturelle (multilingue)", "version": "0.0.0", "termsOfService": "https://services.istex.fr/", "contact": { From fce88a8e3c09aee29e835478731580941a615318 Mon Sep 17 00:00:00 2001 From: leogail Date: Mon, 16 Mar 2026 14:55:07 +0100 Subject: [PATCH 5/9] fix(text-lucene): now, there is the expand id everytime --- services/text-lucene/tests.hurl | 19 +++++++++++++++++++ services/text-lucene/v1/istex-search.ini | 7 +++++-- services/text-lucene/v1/istex-search.py | 5 ++--- 3 files changed, 26 insertions(+), 5 deletions(-) create mode 100644 services/text-lucene/tests.hurl diff --git a/services/text-lucene/tests.hurl b/services/text-lucene/tests.hurl new file mode 100644 index 000000000..7b92b86bc --- /dev/null +++ b/services/text-lucene/tests.hurl @@ -0,0 +1,19 @@ +POST {{host}}/v1/istex-search?indent=true +content-type: application/json +[ + { + "value": "Je veux un corpus d'articles scientifiques publiés post 2020 sur la biodiversité dans la méditerranée nord. Les documents seront en anglais" + }, + { + "value": "no query" + } +] + + +HTTP 200 +[{ + "value": "(title:(\"biodiversity\" \"marine biodiversity\" \"ocean biodiversity\" \"oceanic biodiversity\") OR abstract:(\"biodiversity\" \"marine biodiversity\" \"ocean biodiversity\" \"oceanic biodiversity\") OR subject.value:(\"biodiversity\" \"marine biodiversity\" \"ocean biodiversity\" \"oceanic biodiversity\")) AND (title:(\"north mediterranean\" \"north mediterranean sea\") OR abstract:(\"north mediterranean\" \"north mediterranean sea\") OR subject.value:(\"north mediterranean\" \"north mediterranean sea\")) AND language.raw:(\"eng\") AND genre.raw:\"research-article\" AND publicationDate:[2020 TO *]" +}, +{ + "value": "" +}] \ No newline at end of file diff --git a/services/text-lucene/v1/istex-search.ini b/services/text-lucene/v1/istex-search.ini index a40b47ce4..99c58858a 100644 --- a/services/text-lucene/v1/istex-search.ini +++ b/services/text-lucene/v1/istex-search.ini @@ -13,9 +13,12 @@ post.parameters.0.description = Indenter le JSON résultant post.parameters.0.in = query post.parameters.0.name = indent post.parameters.0.schema.type = boolean -#' # Examples +post.requestBody.content.application/json.example.0.value = Je veux un corpus d'articles scientifiques publiés post 2020 sur la biodiversité dans la méditerranée nord. Les documents seront en anglais +post.requestBody.content.application/json.example.1.value = no query +post.responses.default.content.application/json.example.0.value = (title:("biodiversity" "marine biodiversity" "ocean biodiversity" "oceanic biodiversity") OR abstract:("biodiversity" "marine biodiversity" "ocean biodiversity" "oceanic biodiversity") OR subject.value:("biodiversity" "marine biodiversity" "ocean biodiversity" "oceanic biodiversity")) AND (title:("north mediterranean" "north mediterranean sea") OR abstract:("north mediterranean" "north mediterranean sea") OR subject.value:("north mediterranean" "north mediterranean sea")) AND language.raw:("eng") AND genre.raw:"research-article" AND publicationDate:[2020 TO *] +post.responses.default.content.application/json.example.1.value = [use] plugin = @ezs/spawn @@ -26,7 +29,7 @@ separator = * [expand] path = value -size = 1 +size = 5 [expand/exec] # command should be executable ! diff --git a/services/text-lucene/v1/istex-search.py b/services/text-lucene/v1/istex-search.py index e657d60c9..1d70e2992 100755 --- a/services/text-lucene/v1/istex-search.py +++ b/services/text-lucene/v1/istex-search.py @@ -137,7 +137,7 @@ def main(): user_prompt = json.loads(line) value = user_prompt["value"] if len(value) < 10: - raise Exception("To short request") + raise Exception("Too short request") prompt = construct_llm_prompt(value) output = None @@ -168,8 +168,7 @@ def main(): except Exception as e: sys.stderr.write(f"Unexpected error: {str(e)}") sys.stderr.write("\n") - user_prompt = {"value": ""} - + user_prompt["value"] = "" sys.stdout.write(json.dumps(user_prompt)) sys.stdout.write("\n") From ef23ff4b5e4f6b753dcabc759ea03e4752532e9e Mon Sep 17 00:00:00 2001 From: leogail Date: Mon, 16 Mar 2026 15:17:31 +0100 Subject: [PATCH 6/9] test(text-lucene): update tests (not reproductibles res) --- services/text-lucene/tests.hurl | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/services/text-lucene/tests.hurl b/services/text-lucene/tests.hurl index 7b92b86bc..9bf35f379 100644 --- a/services/text-lucene/tests.hurl +++ b/services/text-lucene/tests.hurl @@ -11,9 +11,11 @@ content-type: application/json HTTP 200 -[{ - "value": "(title:(\"biodiversity\" \"marine biodiversity\" \"ocean biodiversity\" \"oceanic biodiversity\") OR abstract:(\"biodiversity\" \"marine biodiversity\" \"ocean biodiversity\" \"oceanic biodiversity\") OR subject.value:(\"biodiversity\" \"marine biodiversity\" \"ocean biodiversity\" \"oceanic biodiversity\")) AND (title:(\"north mediterranean\" \"north mediterranean sea\") OR abstract:(\"north mediterranean\" \"north mediterranean sea\") OR subject.value:(\"north mediterranean\" \"north mediterranean sea\")) AND language.raw:(\"eng\") AND genre.raw:\"research-article\" AND publicationDate:[2020 TO *]" -}, -{ - "value": "" -}] \ No newline at end of file +Content-Type: application/json +[Asserts] +jsonpath "$[0].value" contains "title:(" +jsonpath "$[0].value" contains "abstract:(" +jsonpath "$[0].value" contains "language.raw:" +jsonpath "$[0].value" matches /^.{100,1100}$/ +jsonpath "$[1].value" exists +jsonpath "$[1].value" matches /^.{0,10}$/ From 27188d43816a883ff1f98058ae394dfae052400a Mon Sep 17 00:00:00 2001 From: leogail Date: Wed, 18 Mar 2026 08:57:38 +0100 Subject: [PATCH 7/9] docs(text-lucene): add modif from review --- services/text-lucene/v1/istex-search.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/services/text-lucene/v1/istex-search.py b/services/text-lucene/v1/istex-search.py index 1d70e2992..66b7b835b 100755 --- a/services/text-lucene/v1/istex-search.py +++ b/services/text-lucene/v1/istex-search.py @@ -56,7 +56,7 @@ def construct_llm_prompt(user_prompt): (title:("electric car" "electrics cars" "electric vehicle" "electrics vehicles") OR abstract:("electric car" "electrics cars" "electric vehicle" "electrics vehicles")) AND publicationDate:[2015 TO *] ``` - 2. Exemple 2 : Si l'utilisateur demande "Trouve tous les articles scientifiques parûts entre 2000 et 2020 sur la mémoire", + 2. Exemple 2 : Si l'utilisateur demande "Trouve tous les articles scientifiques parus entre 2000 et 2020 sur la mémoire", une réponse possible est ``` (title:("memory" "memories" "metamemory" "mémoire" "mémoires" "métamémoire") OR abstract:("memory" "memories" "metamemory" "mémoire" "mémoires" "métamémoire")) AND publicationDate:[2000 TO 2020] AND genre.raw:("research-article") @@ -99,7 +99,7 @@ def construct_llm_prompt(user_prompt): return prompt -def generate_lucen(prompt: str, model_name: str) -> str: +def generate_lucene(prompt: str, model_name: str) -> str: base_url = "https://llm.ilaas.fr/v1" headers = { "Authorization": f"Bearer {api_key}", @@ -109,7 +109,7 @@ def generate_lucen(prompt: str, model_name: str) -> str: "model": model_name, "messages": [{"role": "user", "content": f"{prompt}"}], "stream": False, - "max_tokens": 2000 + "max_tokens": 4096 } response = requests.post( f"{base_url}/chat/completions", @@ -143,10 +143,10 @@ def main(): for retry in range(retries): try: - response = generate_lucen(prompt, model_name=model) + response = generate_lucene(prompt, model_name=model) response = process_llm_response(response) # If model doesn't generate "```", - # this function returns an IdexError. + # this function returns an IndexError. if len(response) < 10: # We consider the response to short to be a lucene eq. raise NoAnswerError("Empty output") From 881e9dce5b8cb83745ad444252e18d01cf3f4399 Mon Sep 17 00:00:00 2001 From: leogail Date: Wed, 18 Mar 2026 08:58:01 +0100 Subject: [PATCH 8/9] release ws-text-lucene@1.0.0 --- services/text-lucene/README.md | 2 +- services/text-lucene/package.json | 2 +- services/text-lucene/swagger.json | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/services/text-lucene/README.md b/services/text-lucene/README.md index e35aecbc4..71a65364e 100644 --- a/services/text-lucene/README.md +++ b/services/text-lucene/README.md @@ -1,4 +1,4 @@ -# ws-text-lucene@0.0.0 +# ws-text-lucene@1.0.0 Assistance à la génération de requête dans Istex Search diff --git a/services/text-lucene/package.json b/services/text-lucene/package.json index dbce2c1f5..ac4a6ced5 100644 --- a/services/text-lucene/package.json +++ b/services/text-lucene/package.json @@ -1,7 +1,7 @@ { "private": true, "name": "ws-text-lucene", - "version": "0.0.0", + "version": "1.0.0", "description": "Assistance à la génération de requête dans Istex Search", "repository": { "type": "git", diff --git a/services/text-lucene/swagger.json b/services/text-lucene/swagger.json index 50f0120a0..5e4381b06 100644 --- a/services/text-lucene/swagger.json +++ b/services/text-lucene/swagger.json @@ -3,7 +3,7 @@ "info": { "title": "text-lucene - Assistance à la génération de requête dans Istex Search", "description": "Génère une requête en syntaxe Lucene à partir d'une requête en langue naturelle (multilingue)", - "version": "0.0.0", + "version": "1.0.0", "termsOfService": "https://services.istex.fr/", "contact": { "name": "Inist-CNRS", From 71217b695d57126a14a123f207cac2fba8830cfd Mon Sep 17 00:00:00 2001 From: leogail Date: Wed, 18 Mar 2026 09:49:15 +0100 Subject: [PATCH 9/9] chore(text-lucene): update port --- services/text-lucene/swagger.json | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/services/text-lucene/swagger.json b/services/text-lucene/swagger.json index 5e4381b06..6f18f8f9c 100644 --- a/services/text-lucene/swagger.json +++ b/services/text-lucene/swagger.json @@ -15,9 +15,9 @@ "x-comment": "Will be automatically completed by the ezs server." }, { - "url": "http://vptdmservices.intra.inist.fr:49225/", + "url": "http://vptdmservices.intra.inist.fr:49376/", "description": "Latest version for production", - "#DISABLED#x-profil": "Standard" + "x-profil": "Standard" } ], "tags": [ @@ -30,4 +30,4 @@ } } ] -} \ No newline at end of file +}