"""Typed questions for each demo case. A case is a dict with: - questions(): the typed questions sent to the model(s); - summary(answers): turns the model answers into what the page displays; - explain(): the case definition shown in "Comment cet exemple est construit"; - models: question key -> model alias (see MODELS in server.py), "typed" if absent. """ # One choice mixes every kind of non-sales email with the lead levels, so the # model compares "scam" against "perfect lead" directly. A single "prospect" # option against the other kinds misread urgent, detailed requests as spam. # With 11+ options this checkpoint's calibration is also sharper. KIND = { "small talk": ("Salutation ou message personnel", "hello, how are you, or a personal message"), "spam": ("Spam", "prize, lottery, phishing"), "job application": ("Candidature", "the sender wants a job or internship"), "vendor pitch": ("Démarchage fournisseur", "the sender is a vendor promoting its own product or agency to us"), "newsletter": ("Newsletter", "mass mailing, marketing content"), "student": ("Étudiant", "questions for studies, no project"), "cold lead": ("Prospect", "a company, but no real project"), "weak lead": ("Prospect", "a vague wish, no budget, no timing"), "average lead": ("Prospect", "a real project, budget and timing not confirmed"), "good lead": ("Prospect", "clear need, planned budget and a deadline"), "perfect lead": ("Prospect", "decision maker, approved budget, urgent problem, firm deadline"), } LEADS = [key for key in KIND if key.endswith(" lead")] # On the test emails, real requests scored 43% or more, everything else 35% or less. LEAD_THRESHOLD = 0.4 # Random letters were classified as leads by the choice above (the model has # to pick something), so they are rejected by a separate coherence check. COHERENCE = "The text is written in meaningful, coherent sentences." COHERENCE_THRESHOLD = 0.25 # Each level describes what the email must contain: bare labels ("good", # "excellent") separated the examples much less. LEVELS = [ "not a sales inquiry", "a contact with no project", "a vague wish, nothing planned", "a real project, but no budget and no date", "a clear need, a planned budget and a deadline", "the decision maker, an approved budget, an urgent problem and a firm deadline within weeks", ] # Shown as a breakdown only; phrased as facts about the email because vaguer # questions returned high probabilities on off-topic text. SIGNALS = { "budget": ("Budget", "The email explicitly mentions a budget amount or an approved budget for the project."), "authority": ("Décideur", "The sender is an executive who can approve the purchase: CEO, CFO, managing " "director, founder, owner or head of department."), "need": ("Besoin", "The email describes a concrete business project or problem the sender wants us " "to solve."), "timing": ("Timing", "The email mentions a deadline or a start date within the next few months."), } def prospect_questions(): questions = { "kind": {"type": "choice", "instructions": "How should our sales team classify this incoming email?", "criteria": {key: text for key, (_, text) in KIND.items()}}, "quality": {"type": "score", "instructions": "How qualified is the sender of this email as a sales lead?", "criteria": LEVELS}, } questions["coherent"] = {"type": "noul", "instructions": COHERENCE} for key, (_, text) in SIGNALS.items(): questions[key] = {"type": "noul", "instructions": text} return questions def prospect_summary(answers): kinds = answers["kind"]["probabilities"] relevance = sum(kinds[key] for key in LEADS) # "score" is the expected zero-based rubric level. quality = answers["quality"]["score"] / (len(LEVELS) - 1) coherent = answers["coherent"]["noul"] >= COHERENCE_THRESHOLD is_lead = coherent and relevance >= LEAD_THRESHOLD if is_lead: kind = "Prospect" elif not coherent: kind = "Texte incompréhensible" else: kind = KIND[max((k for k in kinds if k not in LEADS), key=kinds.get)][0] return { "score": round(quality * 100) if is_lead else 0, "relevance": relevance, "kind": kind, "details": { "kinds": kinds, "coherence": answers["coherent"]["noul"], "levels": list(answers["quality"]["probabilities"].values()), }, "signals": [ {"label": label, "p": answers[key]["noul"]} for key, (label, _) in SIGNALS.items() ], } def prospect_explain(): """The case definition, for the page that explains how the example is built.""" questions = prospect_questions() return { "kind": { "instructions": questions["kind"]["instructions"], "options": [{"key": key, "label": label, "description": text, "lead": key in LEADS} for key, (label, text) in KIND.items()], "threshold": LEAD_THRESHOLD, }, "coherence": {"instructions": COHERENCE, "threshold": COHERENCE_THRESHOLD}, "quality": {"instructions": questions["quality"]["instructions"], "levels": LEVELS}, "signals": [{"label": label, "instructions": text} for label, text in SIGNALS.values()], } # --------------------------------------------------------------------------- # Case 2: review a cold sales email before sending it. # Laya cannot grade a whole email reliably (the best example was labelled # "generic brochure"), but it detects a few concrete features well, so this # case is a checklist rather than a score. # --------------------------------------------------------------------------- # Defect detectors. The general English checkpoint separates these far better # than typed-decisions (margins of 52 to 66 points on the test emails). CHECKS = { "result": { "label": "Résultat chiffré", "instructions": "The email contains a percentage or a number describing a result " "obtained for another client.", "good_when": True, "tip": "Ajoute un résultat obtenu pour un client similaire, chiffré (%, €, temps gagné).", }, "pressure": { "label": "Pas de pression", "instructions": "The email threatens the recipient or says the offer expires today.", "good_when": False, "tip": "Retire l'urgence artificielle et les menaces : elles font fuir.", }, "apology": { "label": "Pas d'excuses", "instructions": "The email apologizes or sounds hesitant.", "good_when": False, "tip": "Assume ta démarche : pas d'excuses, pas de « peut-être » ni de « si vous avez le temps ».", }, } CHECK_THRESHOLD = 0.5 # Next step: a choice works better than yes/no statements, which all failed. NEXT_STEP = { "nothing": ("Aucune demande", "no request at all"), "website": ("Visiter un site", "visit a website or read a brochure"), "opinion": ("Donner son avis", "reply or give an opinion"), "call now": ("Appeler ou acheter tout de suite", "call immediately or buy now"), "meeting": ("Un rendez-vous, sans date", "agree to a meeting, without a date"), "meeting on a date": ("Un créneau précis", "a short call or meeting on a proposed day, such as Thursday or Friday"), } NEXT_STEP_GOOD = "meeting on a date" NEXT_STEP_TIP = "Termine par un créneau précis : « Seriez-vous disponible 15 minutes jeudi ou vendredi ? »" # Only used for a warning: this gate does not separate vague or timid sales # emails from off-topic text well enough to block the checks. As in case 1, a # choice works where the equivalent yes/no statement never fired. TEXT_KIND = { "sales email": "an email that presents a product or service to a potential client", "small talk": "greetings or a personal message to a friend", "other text": "recipe, weather, notes or any other text", "gibberish": "random letters or meaningless words", } SALES_THRESHOLD = 0.12 def review_questions(): questions = {key: {"type": "noul", "instructions": c["instructions"]} for key, c in CHECKS.items()} questions["next_step"] = {"type": "choice", "instructions": "What does the email ask the recipient to do?", "criteria": {k: text for k, (_, text) in NEXT_STEP.items()}} questions["text_kind"] = {"type": "choice", "instructions": "What is this text?", "criteria": TEXT_KIND} questions["coherent"] = {"type": "noul", "instructions": COHERENCE} return questions def review_summary(answers): checks = [] for key, c in CHECKS.items(): p = answers[key]["noul"] ok = (p >= CHECK_THRESHOLD) == c["good_when"] checks.append({"label": c["label"], "ok": ok, "p": p, "tip": None if ok else c["tip"]}) steps = answers["next_step"]["probabilities"] step = max(steps, key=steps.get) step_ok = step == NEXT_STEP_GOOD checks.append({"label": "Prochaine étape claire", "ok": step_ok, "p": steps[step], "detail": NEXT_STEP[step][0], "approx": True, "tip": None if step_ok else NEXT_STEP_TIP}) if answers["coherent"]["noul"] < COHERENCE_THRESHOLD: warning = "Texte incompréhensible : les contrôles n'ont pas de sens ici." elif answers["text_kind"]["probabilities"]["sales email"] < SALES_THRESHOLD: warning = "Ce texte ne ressemble pas à un email commercial." else: warning = None return { "passed": sum(c["ok"] for c in checks), "total": len(checks), "checks": checks, "warning": warning, "details": {"steps": steps, "text_kind": answers["text_kind"]["probabilities"], "coherence": answers["coherent"]["noul"]}, } def review_explain(): return { "checks": [{"key": k, "label": c["label"], "instructions": c["instructions"], "good_when": c["good_when"], "tip": c["tip"], "model": "english"} for k, c in CHECKS.items()], "threshold": CHECK_THRESHOLD, "next_step": {"instructions": review_questions()["next_step"]["instructions"], "options": [{"key": k, "label": label, "description": text, "good": k == NEXT_STEP_GOOD} for k, (label, text) in NEXT_STEP.items()], "model": "typed"}, "sales": {"instructions": "What is this text?", "options": TEXT_KIND, "threshold": SALES_THRESHOLD}, "coherence": {"instructions": COHERENCE, "threshold": COHERENCE_THRESHOLD}, } # --------------------------------------------------------------------------- # Case 3: the Sorting Hat. Pure choice questions, which Laya handles best. # --------------------------------------------------------------------------- HOUSES = { "gryffindor": ("Gryffondor", "brave, daring, adventurous"), "ravenclaw": ("Serdaigle", "intelligent, curious, loves learning"), "hufflepuff": ("Poufsouffle", "loyal, patient, kind, hard-working"), "slytherin": ("Serpentard", "ambitious, cunning, wants power"), } # Short generic descriptions ("strategic leader") attracted every profile; each # option now says what the person does during the heist. ROLES = { "the brain": ("Le cerveau", "the one who plans the heist in every detail"), "the hacker": ("Le hacker", "breaks into computers and security systems"), "the driver": ("Le pilote", "drives the getaway car at full speed"), "the smooth talker": ("Le beau parleur", "charms and distracts the guards with words"), "the muscle": ("Les gros bras", "strong and fearless, handles the physical danger"), "the infiltrator": ("L'infiltré", "quietly gets a job inside the bank months before"), "the lookout": ("Le guetteur", "watches the street and warns the team"), "the artist": ("Le faussaire", "forges paintings and fake documents"), } LANGUAGES = { "Rust": "obsessed with safety, rigor and performance", "Python": "pragmatic, wants results quickly with little code", "Haskell": "loves mathematics, theory and elegance", "C": "old-school, likes total control", "Go": "simple, efficient, no nonsense", "Java": "structured, disciplined, loves rules and processes", "Ruby": "joyful, artistic, expressive", "JavaScript": "improvises, chaotic, does a bit of everything", "Bash": "automates their life with small scripts", "SQL": "organized, loves order and tables", "PHP": "a survivor, practical, underestimated", "Scratch": "playful, a beginner at heart, loves colorful blocks", } SORTING = { "house": ("Which Hogwarts house fits this person best?", {k: text for k, (_, text) in HOUSES.items()}), "role": ("In a heist movie crew, which role would this person play?", {k: text for k, (_, text) in ROLES.items()}), "language": ("If this person were a programming language, which one would they be?", LANGUAGES), } LABELS = { "house": {k: label for k, (label, _) in HOUSES.items()}, "role": {k: label for k, (label, _) in ROLES.items()}, "language": {k: k for k in LANGUAGES}, } def sorting_questions(): return {key: {"type": "choice", "instructions": instructions, "criteria": criteria} for key, (instructions, criteria) in SORTING.items()} def sorting_summary(answers): result = {} for key in SORTING: probs = answers[key]["probabilities"] ranked = sorted(probs, key=probs.get, reverse=True) result[key] = [{"key": k, "label": LABELS[key][k], "p": probs[k]} for k in ranked] return result def sorting_explain(): return {key: {"instructions": instructions, "options": [{"key": k, "label": LABELS[key][k], "description": text} for k, text in criteria.items()]} for key, (instructions, criteria) in SORTING.items()} CASES = { "prospect": {"questions": prospect_questions, "summary": prospect_summary, "explain": prospect_explain, "models": {}}, "review": {"questions": review_questions, "summary": review_summary, "explain": review_explain, "models": {key: "english" for key in CHECKS}}, "sorting": {"questions": sorting_questions, "summary": sorting_summary, "explain": sorting_explain, "models": {}}, }