| 1 |
/** |
| 2 |
* Parse the plain-text day-by-day format emitted by the `trip-itinerary` |
| 3 |
* AI task into the shape TripForm's `itinerary_days` state expects. |
| 4 |
* |
| 5 |
* The LLM is instructed (in the prompt template) to emit: |
| 6 |
* |
| 7 |
* Day 1: Arrive in Kathmandu |
| 8 |
* Welcome briefing at the hotel, group dinner, and an early night. |
| 9 |
* |
| 10 |
* Day 2: Fly to Lukla, trek to Phakding |
| 11 |
* ... |
| 12 |
* |
| 13 |
* In practice models sometimes drift — they may emit "Day 1 -" instead of |
| 14 |
* "Day 1:", wrap with markdown bold, or sneak in a leading preamble. The |
| 15 |
* parser tolerates these by: |
| 16 |
* - matching `^Day N` with a flexible separator (`:`, `-`, `–`, `—`) |
| 17 |
* - stripping leading/trailing whitespace + common markdown chars |
| 18 |
* - dropping anything before the first valid `Day N` block |
| 19 |
* - de-duping consecutive blank lines inside descriptions |
| 20 |
*/ |
| 21 |
|
| 22 |
export interface ParsedDay { |
| 23 |
day: number; |
| 24 |
day_title: string; |
| 25 |
description: string; |
| 26 |
} |
| 27 |
|
| 28 |
/** |
| 29 |
* Match a "Day N: title" line, tolerant of LLM drift: |
| 30 |
* |
| 31 |
* - optional leading `#`s (markdown header) |
| 32 |
* - optional leading `*`s (markdown bold) |
| 33 |
* - "Day" / "day" / "DAY" |
| 34 |
* - `:`, `-`, `–` (en dash), `—` (em dash), or nothing-then-newline as separator |
| 35 |
* - optional trailing `*`s |
| 36 |
* |
| 37 |
* Plain trailing whitespace + control chars are tolerated. |
| 38 |
* |
| 39 |
* NOTE: trailing-asterisk group is `\**?` (ZERO-or-more, lazy), not |
| 40 |
* `\*+?` (one-or-more) — when the model emits a plain "Day 1: Arrive |
| 41 |
* in Kathmandu" with no markdown decoration, a one-or-more match |
| 42 |
* silently rejected every line and the wizard reported "0 days |
| 43 |
* drafted" even though the prose was perfectly valid. |
| 44 |
*/ |
| 45 |
const DAY_HEAD = |
| 46 |
/^\s*(?:#+\s*)?(?:\*+\s*)?day\s+(\d+)\s*(?:[:\-–—]\s*)?(.*?)\s*\**?\s*$/i; |
| 47 |
|
| 48 |
export function parseItineraryText(text: string): ParsedDay[] { |
| 49 |
if (!text) return []; |
| 50 |
const lines = text.replace(/\r\n?/g, "\n").split("\n"); |
| 51 |
|
| 52 |
const days: ParsedDay[] = []; |
| 53 |
let current: ParsedDay | null = null; |
| 54 |
let descriptionBuffer: string[] = []; |
| 55 |
|
| 56 |
const flush = () => { |
| 57 |
if (current) { |
| 58 |
current.description = cleanDescription(descriptionBuffer); |
| 59 |
days.push(current); |
| 60 |
} |
| 61 |
current = null; |
| 62 |
descriptionBuffer = []; |
| 63 |
}; |
| 64 |
|
| 65 |
for (const rawLine of lines) { |
| 66 |
const line = rawLine.trimEnd(); |
| 67 |
const headMatch = DAY_HEAD.exec(line); |
| 68 |
if (headMatch) { |
| 69 |
flush(); |
| 70 |
const dayNum = parseInt(headMatch[1], 10); |
| 71 |
const title = (headMatch[2] || "").replace(/[\*_`]+/g, "").trim(); |
| 72 |
current = { |
| 73 |
day: Number.isFinite(dayNum) && dayNum > 0 ? dayNum : days.length + 1, |
| 74 |
day_title: title, |
| 75 |
description: "", |
| 76 |
}; |
| 77 |
continue; |
| 78 |
} |
| 79 |
if (current) { |
| 80 |
descriptionBuffer.push(line); |
| 81 |
} |
| 82 |
// lines before the first valid "Day N:" header are intentionally dropped |
| 83 |
} |
| 84 |
|
| 85 |
flush(); |
| 86 |
|
| 87 |
// Renumber if the LLM skipped or duplicated day numbers — preserve the |
| 88 |
// order it produced (callers care about positions, not the original |
| 89 |
// numeric labels). |
| 90 |
return days.map((d, idx) => ({ |
| 91 |
day: idx + 1, |
| 92 |
day_title: d.day_title, |
| 93 |
description: d.description, |
| 94 |
})); |
| 95 |
} |
| 96 |
|
| 97 |
function cleanDescription(buffer: string[]): string { |
| 98 |
// Drop leading blank lines, collapse runs of blank lines, strip markdown |
| 99 |
// emphasis that the LLM sometimes wraps individual sentences with. |
| 100 |
const trimmed = buffer.join("\n").replace(/\s+$/g, ""); |
| 101 |
return trimmed |
| 102 |
.replace(/^\s*\n+/g, "") |
| 103 |
.replace(/\n{3,}/g, "\n\n") |
| 104 |
.replace(/^[\*_]+|[\*_]+$/gm, "") |
| 105 |
.trim(); |
| 106 |
} |
| 107 |
|