Skip to content

Data and schema

The support-ticket dataset and the triage schema used by the eval and fine-tuning lessons. The .jsonl files show their first rows.

data/make_tickets.py Python · 117 lines
"""
data/make_tickets.py — the one dataset lessons 07–11 share.
A support-ticket TRIAGE task, the kind an enterprise customer brings on day one:
input : free-text ticket
output: {"category": billing|outage|how-to|abuse, "severity": 1–5, "next_action": str}
Why synthetic? So everyone gets the same numbers and nothing private leaks.
Why this task? It has a checkable answer (so evals are honest), needs structured
output (lesson 07), is cheap to batch (08), and small models improve visibly
with a LoRA (10, 11).
Honest splits: tickets are de-duplicated, the test set never shares a ticket text with
training, and one phrasing per category appears ONLY in the test set, so evals measure
generalisation, not memory. (lessons/17-full-vs-peft-mlx/sft_data_check.py verifies this.)
Writes (deterministic, seed 7):
data/tickets/train.jsonl 800 rows chat format → fine-tuning (Fireworks SFT + MLX LoRA)
data/tickets/valid.jsonl 100 rows chat format → MLX validation loss
data/tickets/test.jsonl 100 rows {"ticket", "category", "severity"} → evals, NEVER trained on
python data/make_tickets.py
"""
import json
import random
from pathlib import Path
OUT = Path(__file__).parent / "tickets"
SYSTEM = ("You triage customer support tickets. Reply with JSON only: "
'{"category": "billing|outage|how-to|abuse", "severity": 1-5, "next_action": "<short>"}')
PRODUCTS = ["the API", "the dashboard", "SSO login", "webhooks", "the EU cluster", "batch jobs", "the CLI"]
REGIONS = ["us-east", "eu-west", "ap-south", "us-west"]
URGENT = ["", "", "", " This is urgent.", " Launch is tomorrow.", " It affects all users.", " We are in production."]
OPENERS = ["", "", "Hi team, ", "Hello, ", "Quick one: ", "Hey, ", "Good morning. "]
CLOSERS = ["", "", " Thanks.", " Please help.", " Account {acct}.", " Ref #{ref}.", " Cheers, {name}."]
NAMES = ["Priya", "Tom", "Wei", "Aisha", "Carlos", "Mei", "Olu", "Sara", "Ken", "Lena"]
TEMPLATES = {
"billing": [
"My card was declined when renewing the plan.", "We were charged twice on the last invoice.",
"Can I get a refund for the unused seats (${amt})?", "The invoice shows the wrong company name.",
"Why did the payment for ${amt} fail?", "Billing page shows a charge I don't recognise.",
],
"outage": [
"{prod} is down in {region}.", "Getting 500 errors from {prod} since 09:00.",
"Requests to {prod} time out after 30s.", "Huge latency spike on {prod} in {region}.",
"{prod} unreachable for our whole team.", "Error rate on {prod} jumped to 40%.",
],
"abuse": [
"Someone is sending spam from an account on your platform.", "We received a phishing email using your logo.",
"I think my API key was stolen — unknown usage overnight.", "An account is scraping our public pages via {prod}.",
"Report fraud: fake invoices sent in your name.", "Abuse report: bot signups from {region}.",
],
"how-to": [
"How do I rotate my API key?", "How to configure {prod} for a second region?",
"Where can I find docs for {prod} rate limits?", "How do I set up SSO with Okta?",
"How to export usage data to CSV?", "Where can I change the default model?",
],
}
BASE_SEV = {"outage": 4, "abuse": 4, "billing": 3, "how-to": 2}
ACTION = {"billing": "route to billing; verify charge", "outage": "page on-call; check status page",
"abuse": "route to trust & safety; suspend key if confirmed", "how-to": "reply with docs link"}
def make(rng: random.Random, held_out: bool) -> dict:
"""held_out=True uses only each category's LAST template (never seen in training)."""
cat = rng.choice(list(TEMPLATES))
pool = TEMPLATES[cat][-1:] if held_out else TEMPLATES[cat][:-1]
text = rng.choice(pool).format(prod=rng.choice(PRODUCTS), region=rng.choice(REGIONS),
amt=rng.choice([49, 120, 480, 900, 2400]))
urgency = rng.choice(URGENT)
close = rng.choice(CLOSERS).format(acct=rng.randint(10000, 99999), ref=rng.randint(1000, 9999), name=rng.choice(NAMES))
sev = min(5, BASE_SEV[cat] + (1 if urgency else 0))
return {"ticket": rng.choice(OPENERS) + text + urgency + close, "category": cat, "severity": sev,
"next_action": ACTION[cat]}
def as_chat(r: dict) -> dict:
answer = {"category": r["category"], "severity": r["severity"], "next_action": r["next_action"]}
return {"messages": [{"role": "system", "content": SYSTEM},
{"role": "user", "content": r["ticket"]},
{"role": "assistant", "content": json.dumps(answer)}]}
def main() -> None:
rng = random.Random(7)
seen: set[str] = set()
def unique(n: int, held_out: bool) -> list[dict]:
out = []
while len(out) < n:
r = make(rng, held_out)
if r["ticket"] not in seen: # de-duplicate across ALL splits
seen.add(r["ticket"])
out.append(r)
return out
train_pool = unique(970, held_out=False)
splits = {
"train": train_pool[:800],
"valid": train_pool[800:900],
# test = 70 unseen tickets in familiar phrasings + 30 in phrasings never seen in training
"test": train_pool[900:970] + unique(30, held_out=True),
}
rng.shuffle(splits["test"])
OUT.mkdir(exist_ok=True)
for name, rs in splits.items():
with (OUT / f"{name}.jsonl").open("w") as f:
for r in rs:
f.write(json.dumps(as_chat(r) if name != "test" else
{k: r[k] for k in ("ticket", "category", "severity")}) + "\n")
print(f"wrote {OUT / name}.jsonl ({len(rs)} rows)")
if __name__ == "__main__":
main()
data/tickets/test.jsonl JSON Lines · 100 lines · first 3 shown
{"ticket": "Good morning. How do I rotate my API key? This is urgent. Cheers, Carlos.", "category": "how-to", "severity": 3}
{"ticket": "Abuse report: bot signups from us-east. It affects all users. Ref #8731.", "category": "abuse", "severity": 5}
{"ticket": "Hello, I think my API key was stolen \u2014 unknown usage overnight. This is urgent. Thanks.", "category": "abuse", "severity": 5}
data/tickets/train.jsonl JSON Lines · 800 lines · first 3 shown
{"messages": [{"role": "system", "content": "You triage customer support tickets. Reply with JSON only: {\"category\": \"billing|outage|how-to|abuse\", \"severity\": 1-5, \"next_action\": \"<short>\"}"}, {"role": "user", "content": "We received a phishing email using your logo. We are in production. Account 22337."}, {"role": "assistant", "content": "{\"category\": \"abuse\", \"severity\": 5, \"next_action\": \"route to trust & safety; suspend key if confirmed\"}"}]}
{"messages": [{"role": "system", "content": "You triage customer support tickets. Reply with JSON only: {\"category\": \"billing|outage|how-to|abuse\", \"severity\": 1-5, \"next_action\": \"<short>\"}"}, {"role": "user", "content": "Good morning. the API is down in us-west."}, {"role": "assistant", "content": "{\"category\": \"outage\", \"severity\": 4, \"next_action\": \"page on-call; check status page\"}"}]}
{"messages": [{"role": "system", "content": "You triage customer support tickets. Reply with JSON only: {\"category\": \"billing|outage|how-to|abuse\", \"severity\": 1-5, \"next_action\": \"<short>\"}"}, {"role": "user", "content": "Quick one: We were charged twice on the last invoice. Launch is tomorrow. Please help."}, {"role": "assistant", "content": "{\"category\": \"billing\", \"severity\": 4, \"next_action\": \"route to billing; verify charge\"}"}]}
data/tickets/valid.jsonl JSON Lines · 100 lines · first 3 shown
{"messages": [{"role": "system", "content": "You triage customer support tickets. Reply with JSON only: {\"category\": \"billing|outage|how-to|abuse\", \"severity\": 1-5, \"next_action\": \"<short>\"}"}, {"role": "user", "content": "How to export usage data to CSV? We are in production. Account 48975."}, {"role": "assistant", "content": "{\"category\": \"how-to\", \"severity\": 3, \"next_action\": \"reply with docs link\"}"}]}
{"messages": [{"role": "system", "content": "You triage customer support tickets. Reply with JSON only: {\"category\": \"billing|outage|how-to|abuse\", \"severity\": 1-5, \"next_action\": \"<short>\"}"}, {"role": "user", "content": "Hi team, Where can I find docs for the API rate limits? We are in production. Cheers, Aisha."}, {"role": "assistant", "content": "{\"category\": \"how-to\", \"severity\": 3, \"next_action\": \"reply with docs link\"}"}]}
{"messages": [{"role": "system", "content": "You triage customer support tickets. Reply with JSON only: {\"category\": \"billing|outage|how-to|abuse\", \"severity\": 1-5, \"next_action\": \"<short>\"}"}, {"role": "user", "content": "Good morning. Getting 500 errors from the dashboard since 09:00. It affects all users. Ref #3327."}, {"role": "assistant", "content": "{\"category\": \"outage\", \"severity\": 5, \"next_action\": \"page on-call; check status page\"}"}]}
data/triage_schema.json JSON · 10 lines
{
"type": "object",
"properties": {
"category": {"type": "string", "enum": ["billing", "outage", "how-to", "abuse"]},
"severity": {"type": "integer", "minimum": 1, "maximum": 5},
"next_action": {"type": "string", "maxLength": 120}
},
"required": ["category", "severity", "next_action"],
"additionalProperties": false
}