trips/mock/server.js
Greg Pomerantz 3da04c111e Move runtime artifacts out of the source tree to ~/trips
The dev directory now holds source code + compiled binaries only:
  ~/trips/osm/          OSM PBF extracts (was osm/, 1.4 GB)
  ~/trips/tilecache/    mock tile cache (was mock/.tilecache)
  ~/trips/store/mock/   per-trip uploads + chat logs (was mock/store, PII)
  ~/trips/logs/         server.log
  ~/trips/stale-osm-build/  stale source/tarball dup of ~/osm-build
                           (repo osm/build had no binaries; the live OSRM
                           toolchain + graphs already lived in ~/osm-build)

Code now points at the new locations, all overridable via env:
- mock/server.js: TRIPS_HOME (default ~/trips) for tile cache + store
- setup-osrm.sh / setup-osrm-colombia.sh: OSM_DIR (default ~/trips/osm)
- spatial/compose+import.sh: OSM_DIR (default ~/trips/osm)
- .gitignore: rewritten — no artifact paths remain in-tree
2026-09-10 10:03:04 -04:00

586 lines
30 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

// Mock server: static UI + local tile backend (proxy -> OSM, disk-cached)
// + routing backend (proxy -> local OSRM instances; per-trip extracts).
// The UI only ever talks to localhost; swap the upstream for a real
// tileserver-gl / OSRM later without touching the frontend.
const http = require('http');
const fs = require('fs');
const path = require('path');
const os = require('os');
const { execFile } = require('child_process');
const PORT = 8077;
const ROOT = __dirname;
// One upstream OSRM per dataset. `probe` is a short routable pair inside the
// extract, used by /router-status (health checks return Ok for 0-distance
// queries only when the graph has edges there).
const ROUTERS = {
northeast: {
url: process.env.OSRM_NORTHEAST || 'http://localhost:5000',
probe: '-71.06,42.35;-71.07,42.36', // Boston, car extract
},
colombia: {
url: process.env.OSRM_COLOMBIA || 'http://localhost:5003',
// The :5003 extract is a *walking* build (foot.lua) — it cannot produce a
// driving route, so 'car' legs would silently come back as a long walk. The
// :5004 build (car.lua, same cropped OSM) serves taxis/airport transfers.
drivingUrl: process.env.OSRM_COLOMBIA_DRIVE || 'http://localhost:5004',
probe: '-75.5478,10.3954;-75.5400,10.4020', // Cartagena, foot extract
},
};
// LLM backend: local llama.cpp OpenAI-compatible server (see ~/.pi/agent/
// models.json). Model + budget are pinned server-side; the client only ever
// sends messages. Qwen3.8-27B-UD3-Q5-dual: --parallel 2, --ctx-size 131072.
const LLM = {
base: process.env.LLM_BASE || 'http://192.168.3.7:1234',
model: process.env.LLM_MODEL || 'Qwen3.8-27B-UD3-Q5-dual',
ctx: 131072,
maxTokens: +(process.env.LLM_MAX_TOKENS || 8192), // thinking + answer share this budget
};
// Web search backend: self-hosted SearXNG (JSON API). The UI only ever sends
// questions; the engine endpoint stays server-side. This is the third leg of
// the DESIGN.md 3-source blend (OSM ground truth → LLM knowledge → live web).
const SEARXNG = { base: process.env.SEARXNG_BASE || 'http://192.168.3.4:30053' };
// PostGIS spatial service (see ../spatial/): live once the import has run.
const SPATIAL = { url: process.env.SPATIAL_URL || 'http://localhost:5005' };
const routerFor = (u) => ROUTERS[u.searchParams.get('router') || 'northeast'] || ROUTERS.northeast;
const PROFILE = { foot: 'walking', walk: 'walking', bike: 'cycling', cycle: 'cycling', car: 'driving', drive: 'driving', transit: 'driving' };
// Runtime artifacts live OUTSIDE the source tree — the repo holds code +
// built binaries only. Home: ~/trips (override with TRIPS_HOME).
const TRIPS_HOME = process.env.TRIPS_HOME || path.join(os.homedir(), 'trips');
const CACHE = path.join(TRIPS_HOME, 'tilecache');
fs.mkdirSync(CACHE, { recursive: true });
// per-trip uploads (originals + parsed) + chat logs: ~/trips/store/mock/<trip>/
const STORE = path.join(TRIPS_HOME, 'store', 'mock');
fs.mkdirSync(STORE, { recursive: true });
const mime = {
'.html': 'text/html', '.js': 'text/javascript', '.css': 'text/css',
'.json': 'application/json', '.svg': 'image/svg+xml', '.png': 'image/png',
};
// ---- document text extraction (used by /parse-doc) --------------------
// Real email HTML defeats the naive <[^>]+> strip (a ">" inside a quoted
// attribute — e.g. an inline SVG data-URI — truncates the tag), so scan
// char-by-char instead: a tag only ends at an UNQUOTED ">"; comments
// (including MSO conditional comments) and whole style/script/head blocks
// are skipped; block-level tags become line breaks; table cells get " | ".
const ENT = { nbsp: ' ', amp: '&', lt: '<', gt: '>', quot: '"', apos: "'", mdash: '—', ndash: '', hellip: '…', rsquo: '', lsquo: '', rdquo: '”', ldquo: '“', laquo: '«', raquo: '»', copy: '©', reg: '®', trade: '™', times: '×', deg: '°' };
function decodeEntities(s) {
return String(s).replace(/&(#x?[0-9a-f]+|[a-z][a-z0-9]*);/gi, (m, e) => {
if (e[0] === '#') {
try { return String.fromCodePoint(e[1].toLowerCase() === 'x' ? parseInt(e.slice(2), 16) : parseInt(e.slice(1), 10)); }
catch { return ' '; }
}
return Object.prototype.hasOwnProperty.call(ENT, e.toLowerCase()) ? ENT[e.toLowerCase()] : m;
});
}
// extract document text: pdftotext for PDFs, MIME decoding for .eml,
// tag-aware stripping for HTML. cb({ok:false,error}) or
// cb({ok:true,kind,name,text,chars,truncated}) — text capped at 60k chars.
const DOC_CAP = 60000; // well under the 131k-token context; flag if we had to cut
function finishParse(in0, cb) {
const name = String(in0.filename || 'document');
const ext = path.extname(name).toLowerCase();
const ctype = String(in0.mime || '');
if (/^image\//.test(ctype) || ['.png', '.jpg', '.jpeg', '.webp', '.gif', '.bmp'].includes(ext))
return cb({ ok: false, error: 'Screenshots and images arent supported yet (no OCR on this box). Save the booking as a PDF, or copy its text and paste it.' });
if (ext === '.msg')
return cb({ ok: false, error: 'Outlook .msg files cant read here — in your mail app use “Forward as HTML” (or Save As → .eml) and drop that instead.' });
let buf = in0.buf;
if (!buf) {
try { buf = Buffer.from(String(in0.b64 || ''), 'base64'); } catch { return cb({ ok: false, error: 'couldnt read the file data' }); }
if (!buf.length) return cb({ ok: false, error: 'that file looks empty' });
if (buf.length > 16e6) return cb({ ok: false, error: 'file too large (max ~16 MB)' });
}
const finish = (text) => {
text = String(text || '').replace(/\r\n/g, '\n').replace(/\t/g, ' ').replace(/[ \u00a0]{2,}/g, ' ').replace(/\n{3,}/g, '\n\n').trim();
if (!text) return cb({ ok: false, error: 'no readable text found in that file' });
// collapse heavy repetition (boilerplate fare rules etc.): keep the first
// two copies of any line that shows up 10+ times, then a single note
const lines = text.split('\n'), count = {};
lines.forEach(l => { const k = l.trim(); if (k) count[k] = (count[k] || 0) + 1; });
const seen = {}, out = [];
for (const l of lines) {
const k = l.trim();
if (k && count[k] >= 10) {
seen[k] = (seen[k] || 0) + 1;
if (seen[k] > 2) { if (out[out.length - 1] !== '__DUP__' + k) out.push('__DUP__' + k); continue; }
}
out.push(l);
}
text = out.map(l => l.startsWith('__DUP__') ? ' [… ' + count[l.slice(7)] + '× this line repeats — shown above]' : l).join('\n');
try { fs.writeFileSync('/tmp/parse-doc-last.txt', 'file: ' + name + ' (' + buf.length + ' bytes) → ' + text.length + ' chars\n' + '=== extracted text (exactly what the model receives) ===\n' + text); } catch {}
console.log(`[parse-doc] ${name} ${buf.length}B → ${text.length} chars${text.length > DOC_CAP ? ' (TRUNCATED at ' + DOC_CAP + ')' : ''} | preview: ${text.slice(0, 200).replace(/\n/g, ' ⏎ ')}`);
cb({ ok: true, kind: ext.replace('.', '') || 'text', name, text: text.slice(0, DOC_CAP), chars: text.length, truncated: text.length > DOC_CAP });
};
if (ext === '.pdf' || ctype.includes('pdf')) {
const tmp = path.join(os.tmpdir(), 'parse-doc-' + Date.now() + '-' + Math.floor(Math.random() * 1e5) + '.pdf');
fs.writeFile(tmp, buf, (werr) => {
if (werr) return cb({ ok: false, error: 'couldnt write temp file' });
execFile('pdftotext', ['-layout', tmp, '-'], (err, stdout) => {
try { fs.unlinkSync(tmp); } catch {}
finish(err ? '' : stdout);
});
});
return;
}
let text = buf.toString('utf8');
if (ext === '.eml' || looksLikeMime(text)) {
const mimeBody = decodeMime(text);
if (mimeBody != null) text = mimeBody; // not really MIME → treat as plain text
}
const head = text.slice(0, 20000);
const tagDensity = (head.match(/<[a-z][\s\/]/gi) || []).length;
if (ext === '.html' || ext === '.htm' || ctype.includes('html') || /<\s*(html|!doctype|head|body)\b/i.test(head) || tagDensity > 30)
text = htmlToText(text);
finish(text);
}
// JSON body reader: cb(errOrNull, parsedObj)
function readBody(req, cb, limit = 24e6) {
let body = '', done = false;
const finish = (err, obj) => { if (done) return; done = true; cb(err, obj); };
req.on('data', c => {
body += c;
if (body.length > limit) { req.destroy(); finish('too large'); }
});
req.on('end', () => {
if (body.length > limit) return finish('too large');
try { finish(null, JSON.parse(body)); } catch { finish('bad json'); }
});
}
function sendJson(res) {
return j => { res.setHeader('Content-Type', 'application/json'); res.writeHead(200); res.end(JSON.stringify(j)); };
}
function htmlToText(html) {
const BLOCK_END = /^<\/(p|div|tr|table|h[1-6]|li|ul|ol|section|header|footer|blockquote|pre|article|main|form|dl|dd|figure)\s*>$/i;
let out = '', i = 0, n = html.length;
while (i < n) {
const c = html[i];
if (c !== '<') { out += c; i++; continue; }
if (html.startsWith('<!--', i)) { // comment (incl. <!--[if gte mso 9]> … <![endif]-->)
const end = html.indexOf('-->', i + 4);
i = end < 0 ? n : end + 3; continue;
}
if (html.startsWith('<!', i) || html.startsWith('<[', i)) { // doctype / stray conditional tag
const end = html.indexOf('>', i); i = end < 0 ? n : end + 1; continue;
}
const skip = /^<(script|style|head|title)\b/i.exec(html.slice(i, i + 40));
if (skip) { // skip the whole block
const cm = new RegExp('</' + skip[1] + '\\s*>', 'i').exec(html.slice(i));
i = cm ? i + cm.index + cm[0].length : n; continue;
}
let j = i + 1, inQ = null; // find tag end, quote-aware
while (j < n) {
const d = html[j];
if (inQ) { if (d === inQ) inQ = null; }
else if (d === '"' || d === "'") inQ = d;
else if (d === '>') break;
j++;
}
const tag = html.slice(i, j < n ? j + 1 : n);
if (/^<br\b/i.test(tag)) out += '\n';
else if (BLOCK_END.test(tag)) out += '\n';
else if (/^<t[dh]\b/i.test(tag)) out += ' | ';
else out += ' ';
i = j < n ? j + 1 : n;
}
return decodeEntities(out);
}
// ---- email (MIME) decoding for .eml / raw mail -------------------------
function looksLikeMime(t) {
const h = t.slice(0, 6000);
return /MIME-Version:\s*1\.0/im.test(h) || /Content-Transfer-Encoding:\s*(base64|quoted-printable)/im.test(h);
}
const decodeQP = (s) => String(s).replace(/=\r?\n/g, '').replace(/=([0-9a-fA-F]{2})/g, (_, hx) => String.fromCharCode(parseInt(hx, 16)));
// top-level header field (starts at column 0 — mail metadata headers like
// "X-Arc: ...content-type:from:subject:..." embed field names in their
// values, so an unanchored case-insensitive match grabs the wrong one);
// last occurrence wins (real headers come after the metadata lines)
const headerField = (h, name) => {
const re = new RegExp('^' + name + ':\\s*([^\\r\\n]+)', 'gim');
let m, last = null;
while ((m = re.exec(h))) last = m[1];
return last ? last.trim().toLowerCase() : '';
};
function decodeMime(raw) {
// best text part (html preferred, then longest); handles one level of
// nested multipart; returns null if nothing decodable was found
const head = raw.slice(0, 20000);
const bm = (head.match(/^boundary="?([^";\r\n]+)"?/im) || head.match(/boundary="?([^";\r\n]+)"?/i));
const parts = [];
if (bm) {
for (const chunk of raw.split(bm[1])) {
const nm = chunk.match(/\r?\n([\s\S]*?)\r?\n\r?\n([\s\S]*)/);
if (!nm) continue;
const h = nm[1];
const ct = headerField(h, 'Content-Type').split(';')[0];
const ce = headerField(h, 'Content-Transfer-Encoding');
const body = nm[2].replace(/\n--\s*$/, '');
if (body.trim()) parts.push({ ct, ce, body });
}
} else {
const nm = raw.match(/\r?\n([\s\S]*?)\r?\n\r?\n([\s\S]*)/);
if (nm) {
const h = nm[1];
parts.push({ ct: headerField(h, 'Content-Type').split(';')[0], ce: headerField(h, 'Content-Transfer-Encoding'), body: nm[2] });
}
}
const dec = [];
for (const p of parts) {
if (/^multipart\//.test(p.ct)) { // nested multipart — recurse
const inner = decodeMime(p.body);
if (inner) dec.push({ html: /<\s*(html|body|div|table|p)\b/i.test(inner.slice(0, 2000)), len: inner.length, body: inner });
continue;
}
if (!/text\/(html|plain)/.test(p.ct)) continue;
let body = p.body;
try {
if (p.ce === 'base64') body = Buffer.from(body.replace(/\s+/g, ''), 'base64').toString('utf8');
else if (p.ce === 'quoted-printable') body = decodeQP(body);
} catch { continue; }
if (body.trim()) dec.push({ html: p.ct.includes('html'), len: body.length, body });
}
dec.sort((a, b) => (b.html - a.html) || (b.len - a.len));
if (dec.length) return dec[0].body;
// last resort: brute-force the biggest base64 block (mail clients vary in
// boundary quoting, charset headers, etc.)
const blocks = raw.match(/[A-Za-z0-9+\/\n=]{400,}/g) || [];
for (const b of blocks.slice(0, 12)) {
try {
const d = Buffer.from(b.replace(/\s+/g, ''), 'base64').toString('utf8');
if (d.length > 200 && /[a-z]{4,}/i.test(d))
dec.push({ html: /<\s*(html|body|div|table|p)\b/i.test(d.slice(0, 2000)), len: d.length, body: d });
} catch {}
}
dec.sort((a, b) => (b.html - a.html) || (b.len - a.len));
return dec.length ? dec[0].body : null;
}
http.createServer((req, res) => {
const u = new URL(req.url, 'http://x');
// ---- routing backend (proxy -> local OSRM) ------------------------
// /route?mode=foot&points=lng,lat;lng,lat[;...]
// /table?mode=foot&points=lng,lat;lng,lat[;...]
if (u.pathname === '/route' || u.pathname === '/table') {
const kind = u.pathname.slice(1);
const mode = u.searchParams.get('mode') || 'foot';
const points = u.searchParams.get('points') || '';
const profile = PROFILE[mode] || 'driving';
let osrmPath = `/${kind}/v1/${profile}/${encodeURIComponent(points)}`;
if (kind === 'route') osrmPath += '?overview=full&alternatives=false&steps=false';
else {
// table: durations from the first point (the anchor) to every point
const n = points.split(';').filter(Boolean).length;
const targets = Array.from({ length: n }, (_, i) => i).join(',');
osrmPath += `?annotations=duration,distance&sources=0&targets=${targets}`;
}
const rt = routerFor(u);
const up = (profile === 'driving' && rt.drivingUrl) ? rt.drivingUrl : rt.url;
const fetchUp = fetch(up + osrmPath).then(r => r.text()).catch(() => null);
fetchUp.then(body => {
res.setHeader('Access-Control-Allow-Origin', '*');
res.setHeader('Content-Type', 'application/json');
if (body == null) { res.writeHead(502); return res.end(JSON.stringify({ code: 'ProxyError' })); }
res.writeHead(200); res.end(body);
});
return;
}
// routing availability probe (so the UI can label router vs estimate)
if (u.pathname === '/router-status') {
const rt = routerFor(u);
fetch(rt.url + '/route/v1/walking/' + rt.probe + '?overview=false').then(r => r.text())
.then(t => { let ok = false; try { ok = JSON.parse(t).code === 'Ok' && JSON.parse(t).routes?.[0]?.distance > 0; } catch (e) {}
res.setHeader('Content-Type', 'application/json'); res.writeHead(200); res.end(JSON.stringify({ router: ok, osrm: rt.url, key: rt === ROUTERS.northeast ? 'northeast' : 'colombia' })); })
.catch(() => { res.setHeader('Content-Type', 'application/json'); res.writeHead(200); res.end(JSON.stringify({ router: false, osrm: rt.url })); });
return;
}
// ---- tile backend -------------------------------------------------
if (u.pathname.startsWith('/tiles/')) {
const parts = u.pathname.split('/');
const [z, x, y] = [parts[2], parts[3], (parts[4] || '').replace(/\.png$/, '')];
if (!/^\d+$/.test(z) || !/^\d+$/.test(x) || !/^\d+$/.test(y)) { res.writeHead(400); return res.end(); }
const file = path.join(CACHE, `${z}/${x}`, `${y}.png`);
const upstream = `https://tile.openstreetmap.org/${z}/${x}/${y}.png`;
res.setHeader('Access-Control-Allow-Origin', '*');
res.setHeader('Content-Type', 'image/png');
res.setHeader('Cache-Control', 'public, max-age=86400');
if (fs.existsSync(file)) { res.writeHead(200); fs.createReadStream(file).pipe(res); return; }
fs.mkdirSync(path.dirname(file), { recursive: true });
const fail = (code) => { if (!res.headersSent) { res.writeHead(code); res.end(); } else res.destroy(); };
// buffer the whole tile (≤100 KB), write the cache file, then reply —
// no stream-splitting races with a client that reads fast
fetch(upstream, { headers: { 'User-Agent': 'mapmock-dev/0.1 (local tile proxy)', 'Referer': `http://localhost:${PORT}/` } })
.then(async r => {
if (!r.ok) { return fail(r.status); }
const buf = Buffer.from(await r.arrayBuffer());
if (buf.length < 20) { return fail(502); } // empty/garbage tile
try { fs.writeFileSync(file, buf); } catch (e) {}
if (res.writableEnded) return;
res.writeHead(200); res.end(buf);
})
.catch(() => fail(502));
return;
}
// ---- LLM backend (proxy -> local llama.cpp server) ----------------
// GET /llm-status — availability probe (mirrors /router-status)
if (u.pathname === '/llm-status') {
fetch(LLM.base + '/v1/models').then(r => r.json())
.then(j => {
const m = (j.data || []).find(x => x.id === LLM.model);
res.setHeader('Access-Control-Allow-Origin', '*');
res.setHeader('Content-Type', 'application/json');
res.writeHead(200);
res.end(JSON.stringify({
llm: !!(m && m.status && m.status.value === 'loaded'),
model: LLM.model, base: LLM.base, ctx: LLM.ctx,
status: (m && m.status && m.status.value) || 'unknown',
}));
})
.catch(() => {
res.setHeader('Access-Control-Allow-Origin', '*');
res.setHeader('Content-Type', 'application/json');
res.writeHead(200); res.end(JSON.stringify({ llm: false, model: LLM.model, base: LLM.base, status: 'unreachable' }));
});
return;
}
// POST /llm-chat — body { messages, max_tokens?, temperature? }.
// Non-streaming; the model is a reasoning model, so the server-side budget
// must cover hidden thinking plus the visible answer.
if (u.pathname === '/llm-chat' && req.method === 'POST') {
let body = '';
req.on('data', c => { body += c; if (body.length > 2e6) req.destroy(); });
req.on('end', () => {
let in0;
try { in0 = JSON.parse(body); } catch { res.writeHead(400); return res.end('bad json'); }
const payload = {
model: LLM.model,
messages: Array.isArray(in0.messages) ? in0.messages : [],
max_tokens: in0.max_tokens || LLM.maxTokens,
temperature: in0.temperature ?? 0.7,
stream: false,
};
res.setHeader('Access-Control-Allow-Origin', '*');
res.setHeader('Content-Type', 'application/json');
fetch(LLM.base + '/v1/chat/completions', {
method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify(payload),
}).then(async r => {
const t = await r.text();
res.writeHead(r.ok ? 200 : 502);
res.end(t);
}).catch(() => {
res.writeHead(502);
res.end(JSON.stringify({ error: 'LLM server unreachable at ' + LLM.base }));
});
});
return;
}
// ---- web search (proxy -> SearXNG) --------------------------------
// GET /search-status — availability probe (HEAD on the root; it does NOT
// touch the upstream engines, unlike a real query)
if (u.pathname === '/search-status') {
fetch(SEARXNG.base + '/', { method: 'HEAD', signal: AbortSignal.timeout(4000) })
.then(r => {
res.setHeader('Access-Control-Allow-Origin', '*');
res.setHeader('Content-Type', 'application/json');
res.writeHead(200);
res.end(JSON.stringify({ web: r.ok, base: SEARXNG.base }));
})
.catch(() => {
res.setHeader('Access-Control-Allow-Origin', '*');
res.setHeader('Content-Type', 'application/json'); res.writeHead(200);
res.end(JSON.stringify({ web: false, base: SEARXNG.base }));
});
return;
}
// GET /search?q=…&categories=…&count=… — normalized result list with
// provenance (url + engine). SearXNG's JSON API wants format=json; the
// proxy caps results and strips the noise fields the UI doesn't render.
if (u.pathname === '/search') {
const q = u.searchParams.get('q') || '';
const count = Math.min(10, Math.max(1, +(u.searchParams.get('count') || 5)));
const cats = u.searchParams.get('categories') || 'general';
if (!q.trim()) { res.writeHead(400, { 'Content-Type': 'application/json' }); return res.end(JSON.stringify({ ok: false, error: 'q is required' })); }
const up = `${SEARXNG.base}/search?q=${encodeURIComponent(q)}&format=json&categories=${encodeURIComponent(cats)}&pageno=1`;
const t0 = Date.now();
fetch(up, { signal: AbortSignal.timeout(20000) }).then(async r => {
if (!r.ok) throw new Error('searxng http ' + r.status);
const j = await r.json();
const results = (j.results || []).slice(0, count).map(x => ({
title: String(x.title || ''), url: String(x.url || ''),
source: String(x.engine || ''), snippet: String(x.content || ''),
})).filter(x => x.url);
console.log(`[search] ${JSON.stringify(q).slice(0, 80)}${results.length} results in ${Date.now() - t0}ms`);
res.setHeader('Access-Control-Allow-Origin', '*');
res.setHeader('Content-Type', 'application/json');
res.writeHead(200);
res.end(JSON.stringify({ ok: true, query: q, count: results.length, results }));
}).catch(e => {
res.setHeader('Access-Control-Allow-Origin', '*');
res.setHeader('Content-Type', 'application/json'); res.writeHead(200);
res.end(JSON.stringify({ ok: false, error: 'search failed: ' + e.message }));
});
return;
}
// ---- spatial backend (proxy -> PostGIS spatiald) -------------------
// /spatial-status probes /health; /spatial/<path> forwards verbatim
// (query string included) — /near, /nearest, /corridor.
if (u.pathname === '/spatial-status' || u.pathname.startsWith('/spatial/')) {
const up = u.pathname === '/spatial-status'
? SPATIAL.url + '/health'
: SPATIAL.url + u.pathname.slice('/spatial'.length); // keep query string
const isProbe = u.pathname === '/spatial-status';
fetch(up, { signal: AbortSignal.timeout(15000) }).then(async r => {
const t = await r.text();
res.setHeader('Access-Control-Allow-Origin', '*');
res.setHeader('Content-Type', 'application/json');
res.writeHead(200);
if (!isProbe) return res.end(t); // forward the service body verbatim
let j = {}; try { j = JSON.parse(t); } catch {}
res.end(JSON.stringify({ spatial: r.ok && !!j.ok, url: SPATIAL.url, ...j }));
}).catch(() => {
res.setHeader('Access-Control-Allow-Origin', '*');
res.setHeader('Content-Type', 'application/json'); res.writeHead(200);
if (u.pathname === '/spatial-status') return res.end(JSON.stringify({ spatial: false, url: SPATIAL.url }));
res.end(JSON.stringify({ error: 'spatial backend unreachable at ' + SPATIAL.url }));
});
return;
}
// ---- document ingestion (flight/booking → text) --------------------
// POST /parse-doc — body { filename, mime, b64 }. Returns { ok, kind,
// name, text, chars, truncated }. PDFs go through pdftotext; emails (.eml
// or MIME-looking text) have their text parts decoded; HTML is reduced to
// clean text with htmlToText(); txt/ics/csv/json read directly. Images
// are rejected (no OCR on this box).
if (u.pathname === '/parse-doc' && req.method === 'POST') {
readBody(req, (err, in0) => {
if (err) { res.writeHead(400); return res.end('bad request'); }
finishParse(in0, sendJson(res));
});
return;
}
// ---- per-trip server store ------------------------------------------
// Storage is cheap: for every trip we keep the ORIGINAL uploaded files,
// the extracted text of each, and the full LLM conversation (jsonl).
// store/<trip>/docs/<docId>-<filename> original bytes
// store/<trip>/docs/<docId>-<filename>.txt extracted text + header
// store/<trip>/chat.jsonl {role,text,at} per line
if (u.pathname.startsWith('/trip-store/')) {
const seg = u.pathname.slice('/trip-store/'.length).split('/');
const tripId = decodeURIComponent(seg[0] || '');
if (!/^[a-z0-9][a-z0-9_-]{0,31}$/i.test(tripId)) { res.writeHead(400); return res.end('bad trip id'); }
const dir = path.join(STORE, tripId);
fs.mkdirSync(path.join(dir, 'docs'), { recursive: true });
const json = sendJson(res);
if (req.method === 'GET' && seg[1] === undefined) {
// manifest: docs + chat stats for this trip
const docs = fs.readdirSync(path.join(dir, 'docs'))
.filter(f => !f.endsWith('.txt'))
.map(f => {
const docId = f.split('-')[0];
let name = f.slice(docId.length + 1), originalSize = 0, text = '', chars = 0, truncated = false;
try { originalSize = fs.statSync(path.join(dir, 'docs', f)).size; } catch { originalSize = -1; }
try {
const t = fs.readFileSync(path.join(dir, 'docs', f + '.txt'), 'utf8');
const nl = t.indexOf('\n');
const m = t.slice(0, nl).match(/^file: (.+) \((\d+) bytes\) → (\d+) chars( \(truncated\))?/);
if (m) { name = m[1]; originalSize = +m[2]; chars = +m[3]; truncated = !!m[4]; }
text = t.slice(nl + 1);
} catch {}
const kind = path.extname(name).toLowerCase().replace('.', '') || 'text';
return { docId, name, kind, originalSize, chars, truncated, text };
});
let chatCount = 0, chatBytes = 0;
try {
chatBytes = fs.statSync(path.join(dir, 'chat.jsonl')).size;
chatCount = fs.readFileSync(path.join(dir, 'chat.jsonl'), 'utf8').split('\n').filter(Boolean).length;
} catch {}
return json({ trip: tripId, docs, chat: { count: chatCount, bytes: chatBytes } });
}
if (req.method === 'GET' && seg[1] === 'chat') {
let messages = [];
try {
messages = fs.readFileSync(path.join(dir, 'chat.jsonl'), 'utf8')
.split('\n').filter(Boolean).slice(-100)
.map(l => { try { return JSON.parse(l); } catch { return null; } })
.filter(Boolean);
} catch {}
return json({ messages });
}
if (req.method === 'GET' && seg[1] === 'doc' && seg[2]) {
const docId = seg[2];
if (!/^[a-z0-9_-]+$/i.test(docId)) { res.writeHead(400); return res.end('bad doc id'); }
const f = fs.readdirSync(path.join(dir, 'docs')).find(x => x.startsWith(docId + '-') && !x.endsWith('.txt'));
if (!f) { res.writeHead(404); return res.end('no such doc'); }
let name = f.slice(docId.length + 1), originalSize = 0, text = '';
try { originalSize = fs.statSync(path.join(dir, 'docs', f)).size; } catch {}
try {
const t = fs.readFileSync(path.join(dir, 'docs', f + '.txt'), 'utf8');
const nl = t.indexOf('\n');
const m = t.slice(0, nl).match(/^file: (.+) \((\d+) bytes\) → (\d+) chars/);
if (m) { name = m[1]; originalSize = +m[2]; }
text = t.slice(nl + 1);
} catch {}
const kind = path.extname(name).toLowerCase().replace('.', '') || 'text';
return json({ docId, name, kind, originalSize, text });
}
if (req.method === 'POST' && seg[1] === 'doc') {
readBody(req, (err, in0) => {
if (err) return json({ ok: false, error: err === 'too large' ? 'file too large (max ~16 MB)' : 'bad request' });
const rawName = String(in0.filename || 'document')
.replace(/[\/\\*?"<>|\x00-\x1f]/g, '_').trim().slice(-80) || 'document';
let buf; try { buf = Buffer.from(String(in0.b64 || ''), 'base64'); } catch { return json({ ok: false, error: 'couldnt read the file data' }); }
if (!buf.length) return json({ ok: false, error: 'that file looks empty' });
finishParse({ filename: rawName, mime: in0.mime || '', buf }, (out) => {
if (!out.ok) return json({ ok: false, error: out.error });
const docId = Date.now().toString(36) + Math.random().toString(36).slice(2, 6);
const base = path.join(dir, 'docs', docId + '-' + rawName);
try {
fs.writeFileSync(base, buf);
fs.writeFileSync(base + '.txt', 'file: ' + rawName + ' (' + buf.length + ' bytes) → ' + out.chars + ' chars' + (out.truncated ? ' (truncated)' : '') + '\n=== extracted text ===\n' + out.text);
} catch { return json({ ok: false, error: 'couldnt store the document on the server' }); }
console.log(`[store] ${tripId}/docs/${docId}-${rawName} ${buf.length}B → ${out.chars} chars`);
json({ ok: true, docId, name: rawName, kind: out.kind, text: out.text, chars: out.chars, truncated: out.truncated, originalSize: buf.length });
});
});
return;
}
if (req.method === 'POST' && seg[1] === 'chat') {
readBody(req, (err, in0) => {
if (err) return json({ ok: false });
const role = in0.role === 'assistant' ? 'assistant' : 'user';
const text = String(in0.text || '');
if (!text.trim()) return json({ ok: false });
fs.appendFile(path.join(dir, 'chat.jsonl'), JSON.stringify({ role, text, at: Date.now() }) + '\n', e => json({ ok: !e }));
});
return;
}
res.writeHead(404); return res.end('no such store route');
}
// ---- static -------------------------------------------------------
let p = u.pathname === '/' ? '/index.html' : u.pathname;
const file = path.join(ROOT, p);
if (!file.startsWith(ROOT)) { res.writeHead(403); return res.end(); }
fs.readFile(file, (e, d) => {
if (e) { res.writeHead(404); return res.end('not found'); }
res.writeHead(200, { 'Content-Type': mime[path.extname(file)] || 'application/octet-stream' });
res.end(d);
});
}).listen(PORT, '0.0.0.0', () => console.log(`map mock: http://localhost:${PORT} (tiles: /tiles/{z}/{x}/{y}.png · llm: ${LLM.base} ${LLM.model})`));