Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions k8s/welearn-api/values.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,7 @@ config:
AZURE_API_VERSION: "2024-08-01-preview"
LLM_MODEL_NAME: gpt-oss-120b
LLM_TEMPERATURE: 0.2
TUTOR_SINGLE_PASS: "true"
MISTRAL_LLM_MODEL_NAME: mistral-small-latest
AZURE_BAML_LLM_FRANCE_BASE_URL: https://welearn-mistral.services.ai.azure.com/openai/v1/
AZURE_BAML_LLM_SWEDEN_BASE_URL: https://welearn-mistral-sweden.services.ai.azure.com/openai/v1/
Expand Down
2 changes: 2 additions & 0 deletions src/app/core/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -48,6 +48,8 @@ def get_api_version(self) -> dict:

LLM_MODEL_NAME: str
LLM_TEMPERATURE: float
# old tutor: one structured LLM call instead of the 3-agent chain (False = revert)
TUTOR_SINGLE_PASS: bool = True

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

this needs to be added to the values.yaml in the k8s folder

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Done in 8600ba8

ENV: str

# LANGSMITH / LANGCHAIN TRACING
Expand Down
6 changes: 3 additions & 3 deletions src/app/shared/utils/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -39,9 +39,9 @@ def extract_doc_info(documents: list[ScoredPoint]) -> list[dict]:
"""
return [
{
"title": getattr(doc.payload, "document_title", ""), # type: ignore
"url": getattr(doc.payload, "document_url", ""), # type: ignore
"content": getattr(doc.payload, "slice_content", ""), # type: ignore
"title": doc.payload.get("document_title", ""),
"url": doc.payload.get("document_url", ""),
"content": doc.payload.get("slice_content", ""),
}
for doc in documents
if doc.payload is not None
Expand Down
139 changes: 139 additions & 0 deletions src/app/tests/services/tutor/test_syllabus.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,139 @@
from src.app.tutor.service.models import (
DraftAssessment,
DraftCompetency,
DraftOutcome,
DraftSession,
DraftSyllabus,
)
from src.app.tutor.service.syllabus import (
compute_limits,
detect_syllabus_lang,
render_markdown,
render_references,
split_references,
trim_chatter,
)


def make_draft() -> DraftSyllabus:
return DraftSyllabus(
course_title="Économie circulaire",
description="Desc",
objectives=["O1", "O2", "O3", "O4", "O5"],
outcomes=[
DraftOutcome(text=f"Out {i}", objective_numbers=[1]) for i in range(1, 10)
],
competencies=[
DraftCompetency(text="Travail en équipe", outcome_numbers=[1]),
DraftCompetency(
text="Systémique", greencomp_code="2.1", outcome_numbers=[2]
),
DraftCompetency(text="Non liée", greencomp_code="2.2", outcome_numbers=[]),
DraftCompetency(text="Critique", greencomp_code="2.2", outcome_numbers=[3]),
DraftCompetency(text="Doublon", greencomp_code="2.2", outcome_numbers=[1]),
DraftCompetency(
text="Faux code", greencomp_code="9.9", outcome_numbers=[1]
),
],
assessment=[
DraftAssessment(method="Projet", weight="100%", outcome_numbers=[1, 99])
],
schedule=[DraftSession(topics="A | B", outcome_numbers=[1], class_plan="x\ny")],
)


def test_compute_limits():
assert compute_limits(None).sessions is None
assert compute_limits(None).outcomes == 5 # sized like a semester course
assert compute_limits("4 semaines").outcomes == 3
assert compute_limits("8 semaines").outcomes == 4 # grows with duration
assert compute_limits("8 semaines").objectives == 3
assert compute_limits("12 semaines").sessions == 12
assert compute_limits("30h").sessions == 10
assert compute_limits("1 semestre").sessions == 12
assert compute_limits("40 weeks").outcomes == 7 # ceiling
assert compute_limits("40 weeks").objectives == 4


def test_render_markdown_fr():
md = render_markdown(make_draft(), "fr", compute_limits("12 semaines"))
assert md.startswith("# Économie circulaire")
assert "## 3. Résultats d'apprentissage" in md and "Learning" not in md
assert "**RA5**" in md and "**RA6**" not in md # capped at 5 outcomes
assert "4. O4" not in md # capped at 3 objectives
assert "Non liée" not in md and "Doublon" not in md # unlinked / duplicate code
# official name in the syllabus language, GreenComp listed first
assert "**GreenComp 2.1 – Pensée systémique** : Systémique *(RA2)*" in md
assert "**GreenComp 2.2 – Pensée critique** : Critique" in md
assert md.index("GreenComp 2.1") < md.index("Travail en équipe")
assert "- Faux code" in md # unknown code -> plain competency
assert "RA99" not in md
assert "| 1 | A / B | RA1 | x<br>y |" in md


def test_references_are_deterministic():
refs = render_references(
[
{"title": "Doc A", "url": "https://a.org"},
{"title": "Doc A", "url": "https://a.org"}, # second slice of same doc
{"title": "Doc B", "url": "https://b.org"},
],
"fr",
)
assert refs.startswith("## 7. Références")
assert refs.count("https://a.org") == 1
assert '<a href="https://b.org" target="_blank">[lien]</a>' in refs
assert "Aucune" in render_references([], "fr")
# a resource without title or url is never rendered
assert "Aucune" in render_references(
[{"title": "", "url": "https://c.org"}, {"title": "Doc C", "url": ""}], "fr"
)


def test_split_and_trim():
md = (
"Voici votre syllabus :\n# T\n## 6. Programme\n| a |\n|---|\n"
"Ce syllabus est aligné avec GreenComp.\n## 7. Références\n- Doc <a>x</a>"
)
body, refs = split_references(md)
assert refs == "## 7. Références\n- Doc <a>x</a>"
assert trim_chatter(body) == "# T\n## 6. Programme\n| a |\n|---|"
assert split_references("# T\nno refs") == ("# T\nno refs", "")


def test_detect_syllabus_lang():
assert (
detect_syllabus_lang(render_markdown(make_draft(), "fr", compute_limits(None)))
== "fr"
)
assert (
detect_syllabus_lang(render_markdown(make_draft(), "en", compute_limits(None)))
== "en"
)
assert detect_syllabus_lang("# free text") == "unknown"


def test_generate_syllabus_retries_once_then_raises():
import asyncio

import pytest
from langchain_core.runnables import RunnableConfig, RunnableLambda

from src.app.tutor.service.models import MessageWithResources
from src.app.tutor.service.syllabus import generate_syllabus

answers = [None, make_draft()] # first answer invalid, second valid

class FakeModel:
def with_structured_output(self, schema):
return RunnableLambda(lambda _: answers.pop(0))

msg = MessageWithResources(
lang="fr", content=[], themes=[], summary=[], resources=[], duration=None
)
md = asyncio.run(generate_syllabus(msg, FakeModel(), [], RunnableConfig())) # type: ignore
assert md.startswith("# Économie circulaire") and not answers

answers[:] = [None, None]
with pytest.raises(Exception):
asyncio.run(generate_syllabus(msg, FakeModel(), [], RunnableConfig())) # type: ignore
20 changes: 10 additions & 10 deletions src/app/tests/services/tutor/test_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,18 +42,18 @@ def test_build_system_message_without_optional_params(self):
def test_extract_doc_info(self):
# Create mock documents
doc1 = Mock(spec=ScoredPoint)
doc1.payload = Mock(
document_title="Test Doc 1",
document_url="http://test1.com",
slice_content="Content 1",
)
doc1.payload = {
"document_title": "Test Doc 1",
"document_url": "http://test1.com",
"slice_content": "Content 1",
}

doc2 = Mock(spec=ScoredPoint)
doc2.payload = Mock(
document_title="Test Doc 2",
document_url="http://test2.com",
slice_content="Content 2",
)
doc2.payload = {
"document_title": "Test Doc 2",
"document_url": "http://test2.com",
"slice_content": "Content 2",
}

doc3 = Mock(spec=ScoredPoint)
doc3.payload = None # Test case for None payload
Expand Down
101 changes: 27 additions & 74 deletions src/app/tutor/api/router.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,9 +6,11 @@
BackgroundTasks,
Depends,
File,
HTTPException,
Request,
Response,
UploadFile,
status,
)

from src.app.baml_client.async_client import b, types
Expand All @@ -22,9 +24,6 @@
from src.app.shared.utils.dependencies import get_settings
from src.app.shared.utils.requests import extract_session_cookie
from src.app.shared.utils.utils import get_files_content

# from src.app.tutor.service.agents import TEMPLATES
from src.app.tutor.service.agents import TEMPLATES
from src.app.tutor.service.models import (
CompetencyMappingRequest,
CourseDescriptionRequest,
Expand All @@ -48,7 +47,7 @@
extractor_user_prompt,
summaries_schema,
)
from src.app.tutor.service.tutor import tutor_manager
from src.app.tutor.service.tutor import apply_feedback, tutor_manager
from src.app.utils.logger import logger as utils_logger

logger = utils_logger(__name__)
Expand Down Expand Up @@ -197,7 +196,6 @@ async def tutor_search_extract(
return resp


@with_backoff()
@router.post("/syllabus")
async def create_syllabus(
request: Request,
Expand All @@ -214,14 +212,26 @@ async def create_syllabus(
"query_extracts_count": len(body.extracts),
"documents_count": len(body.documents),
}
results = await tutor_manager(body, lang, settings, trace_context=trace_context)
try:
results = await tutor_manager(body, lang, settings, trace_context=trace_context)
except Exception as e:
logger.error(f"Syllabus generation failed: {e}")
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail={
"message": "Syllabus generation failed, please retry",
"code": "SYLLABUS_GENERATION_FAILED",
},
)

# TODO: handle errors
# the final syllabus is the last item: the only one in single-pass mode,
# the PedagogicalEngineerAgent's in the legacy 3-agent chain
final_syllabus = results[-1]

message_id = await data_collection.register_syllabus_data(
session_id=session_id,
input_data=body,
agent_answer=results[0].content if results else "",
agent_answer=final_syllabus.content,
feature="syllabus_creation",
)

Expand All @@ -233,81 +243,24 @@ async def create_syllabus(
)


feedback_prompt = """
You are a pedagogical engineer and are given a syllabus and a feedback. by the teacher that will teach the course.
Your responsibility is to analyze the syllabus and return an improved version of it in a markdown format. Do not add the backticks and the markdown mention.
It is important to take into account the feedback given by the teacher and to keep the syllabus structure.
The syllabus structure is:
{syllabus_structure}

To be able to do that, the assistant gives you:
- the syllabus of the course
- the feedback given by the teacher
- a list of documents related to the course
- a list of extracts from a document of interest
- the themes that the course is related to

You will respond with the syllabus. Do not provide explanations or notes
"""

feedback_assistant_prompt = """
IMPORTANT: you must follow the syllabus structure given by the system message.
and the respect the format of the original syllabus.
Keep the same language as the original syllabus.

here is the original syllabus:
{syllabus}

take into account the user feedback:
{feedback}

keep the references section with the format <a href="document.url">document.title</a>, references are based on these documents:
{documents}

for more context, here are the extracts of the original document the user sent to build the syllabus from. Extracts:
{extracts}

and the themes extracted from those documents:
{themes}
"""


@with_backoff()
@router.post("/syllabus/feedback")
async def handle_syllabus_feedback(
request: Request,
body: SyllabusFeedback,
chatfactory=Depends(get_chat_service),
data_collection=Depends(get_data_collection_service),
settings: Settings = Depends(get_settings),
):
session_id = extract_session_cookie(request)

messages = [
{
"role": "system",
"content": feedback_prompt.format(syllabus_structure=TEMPLATES),
},
{
"role": "user",
"content": feedback_assistant_prompt.format(
syllabus=body.syllabus[0],
feedback=body.feedback,
documents=body.documents,
extracts="\n".join([extract.summary for extract in body.extracts]),
themes=(", ").join(
[
(", ").join([theme["theme"] for theme in extract.themes])
for extract in body.extracts
]
),
),
},
]

try:
syllabus = await chatfactory.syllabus_feedback_completion(
max_tokens=20000,
messages=messages,
syllabus = await apply_feedback(
body,
settings,
trace_context={
"endpoint": request.url.path,
"feature": "syllabus_feedback",
"session_id": str(session_id) if session_id else None,
},
)

await data_collection.register_syllabus_data(
Expand Down
2 changes: 1 addition & 1 deletion src/app/tutor/domain/template.md
Original file line number Diff line number Diff line change
Expand Up @@ -22,4 +22,4 @@ Course title: Small title for the course
table structure: | week | Topics | Learning Outcomes | Class Plan |
7. References: In this section should be added all sources used to generate the syllabus, including any source documents or WeLearn documents used to construct the syllabus.
IMPORTANT: Do not remove any existent references in this section
Example of references: Document title <a href="document url" _target=blank>[link]</a>
Example of references: Document title <a href="document url" target="_blank">[link]</a>
Loading
Loading