From fb756026fa79e6124a146805092dfac9266161c5 Mon Sep 17 00:00:00 2001 From: Syed Ali Abbas Zaidi <88369802+Syed-Ali-Abbas-Zaidi@users.noreply.github.com> Date: Mon, 28 Sep 2026 11:57:20 +0500 Subject: [PATCH] feat(import): add LinkedIn data export as a resume import source (#3538) * feat(import): add LinkedIn data export as a resume import source LinkedIn's "Get a copy of your data" export ships a ZIP of per-topic CSVs (Profile, Positions, Education, Skills, Languages, Certifications). Reading these directly gives structured data without needing a connected AI provider, unlike the existing PDF/DOCX import path. Also fixes a latent bug found while building this: parseJSONResume (and the new LinkedIn parser) built their result via a shallow spread of the shared `defaultResumeData` singleton, so assigning into `result.sections.x` mutated that singleton in place and leaked section data into the next unrelated import call in the same process. Both now start from a structuredClone. * fix(import): escape LinkedIn text, harden zip parsing and date handling Move escapeHtml and toHtml from the plain-text importer into html.ts and use them for LinkedIn summary, position descriptions and education notes, so CSV text is HTML-escaped and line breaks become paragraphs or bullet lists instead of collapsing into one run-on paragraph. Only an empty end date now marks an entry as ongoing. Date cells that are not "Mon YYYY" are kept verbatim, so a finished role with an unexpected date format no longer reads as "Present". Unzip only the six CSVs the importer reads, matched by exact file name, and reject any of them larger than 5 MB. This avoids inflating the rest of a complete LinkedIn export in the browser and stops Learning_Profile.csv being read as Profile.csv. Map LinkedIn's five language proficiency options onto levels 5 to 1, falling back to parseLevel for anything else. Drop the literal BOM strip, which TextDecoder already handles. The import dialog no longer mentions an AI provider in the loading toast for LinkedIn imports, which are parsed entirely in the browser. Add a regression test for the JSON Resume importer leaking section data through the shared defaultResumeData object, plus LinkedIn tests for HTML escaping, unrecognised end dates, BOM headers, exact file name matching, language levels and oversized entries. --------- Co-authored-by: Amruth Pillai --- apps/web/src/dialogs/resume/import.tsx | 39 ++- apps/web/src/dialogs/resume/import.utils.ts | 9 +- packages/import/package.json | 2 + packages/import/src/html.ts | 26 ++ packages/import/src/json-resume.test.ts | 9 + packages/import/src/json-resume.tsx | 7 +- packages/import/src/linkedin.test.ts | 163 +++++++++++ packages/import/src/linkedin.ts | 284 ++++++++++++++++++++ packages/import/src/plain-text.ts | 23 +- pnpm-lock.yaml | 3 + 10 files changed, 535 insertions(+), 30 deletions(-) create mode 100644 packages/import/src/linkedin.test.ts create mode 100644 packages/import/src/linkedin.ts diff --git a/apps/web/src/dialogs/resume/import.tsx b/apps/web/src/dialogs/resume/import.tsx index bfe2a67dc..a0520ee52 100644 --- a/apps/web/src/dialogs/resume/import.tsx +++ b/apps/web/src/dialogs/resume/import.tsx @@ -57,6 +57,17 @@ const formSchema = z.discriminatedUnion("type", [ { message: "File must be a Microsoft Word document" }, ), }), + z.object({ + type: z.literal("linkedin"), + file: z + .instanceof(File) + .refine( + (file) => file.type === "" || file.type === "application/zip" || file.name.toLowerCase().endsWith(".zip"), + { + message: "File must be a ZIP archive", + }, + ), + }), z.object({ type: z.literal("reactive-resume-json"), file: z @@ -101,6 +112,11 @@ async function detectImportType(file: File): Promise { const isZip = header[0] === 0x50 && header[1] === 0x4b && header[2] === 0x03 && header[3] === 0x04; // "PK\x03\x04" if (isPdf || mime === "application/pdf" || name.endsWith(".pdf")) return "pdf"; + + // Word documents are also ZIPs, so a bare "PK" header is ambiguous. LinkedIn's export is + // only ever named with a .zip extension, so check that first and let it win the tie. + if (name.endsWith(".zip") || mime === "application/zip") return "linkedin"; + if ( isZip || mime === "application/msword" || @@ -144,13 +160,13 @@ export function ImportResumeDialog(_: DialogProps<"resume.import">) { setIsImporting(true); - // A PDF parsed in the browser never touches a provider, so promising one would be a lie. - const isLocalPdf = value.type === "pdf" && !hasUsableProvider; + // A LinkedIn export, or a PDF parsed in the browser, never touches a provider, so promising one would be a lie. + const isLocalParse = value.type === "linkedin" || (value.type === "pdf" && !hasUsableProvider); const toastId = toast.add({ type: "loading", title: t`Importing your resume...`, - description: isLocalPdf + description: isLocalParse ? t`This may take a moment. Please do not close the window or refresh the page.` : t`This may take a few minutes, depending on the response of the AI provider. Please do not close the window or refresh the page.`, }); @@ -195,6 +211,12 @@ export function ImportResumeDialog(_: DialogProps<"resume.import">) { } } + if (value.type === "linkedin") { + const { parseLinkedInExport } = await import("@reactive-resume/import/linkedin"); + const bytes = new Uint8Array(await value.file.arrayBuffer()); + data = parseLinkedInExport(bytes); + } + if (value.type === "docx") { if (isLoadingAiProviders) throw new Error(t`Loading AI providers. Please try again in a moment.`); if (!hasUsableProvider) @@ -315,7 +337,8 @@ export function ImportResumeDialog(_: DialogProps<"resume.import">) { Continue where you left off by importing a resume you built in Reactive Resume or another resume builder. - Supported formats are PDF, Microsoft Word, and JSON files from Reactive Resume or JSON Resume. + Supported formats are PDF, Microsoft Word, a LinkedIn data export, and JSON files from Reactive Resume or + JSON Resume. @@ -401,6 +424,14 @@ export function ImportResumeDialog(_: DialogProps<"resume.import">) { message: "JSON Resume", }), }, + { + value: "linkedin", + textValue: "LinkedIn", + label: t({ + comment: "Import source option for a LinkedIn data export ZIP", + message: "LinkedIn (Data Export)", + }), + }, { value: "pdf", textValue: "PDF", diff --git a/apps/web/src/dialogs/resume/import.utils.ts b/apps/web/src/dialogs/resume/import.utils.ts index 2d88de86e..99aaa4bc4 100644 --- a/apps/web/src/dialogs/resume/import.utils.ts +++ b/apps/web/src/dialogs/resume/import.utils.ts @@ -1,4 +1,11 @@ -export type ImportType = "" | "pdf" | "docx" | "reactive-resume-json" | "reactive-resume-v4-json" | "json-resume-json"; +export type ImportType = + | "" + | "pdf" + | "docx" + | "linkedin" + | "reactive-resume-json" + | "reactive-resume-v4-json" + | "json-resume-json"; export function detectJsonImportType(parsed: unknown): ImportType { if (!parsed || typeof parsed !== "object") return ""; diff --git a/packages/import/package.json b/packages/import/package.json index 813f81827..cac807528 100644 --- a/packages/import/package.json +++ b/packages/import/package.json @@ -5,6 +5,7 @@ "private": true, "exports": { "./json-resume": "./src/json-resume.tsx", + "./linkedin": "./src/linkedin.ts", "./plain-text": "./src/plain-text.ts", "./reactive-resume-json": "./src/reactive-resume-json.tsx", "./reactive-resume-v4-json": "./src/reactive-resume-v4-json.tsx" @@ -20,6 +21,7 @@ "@reactive-resume/resume": "workspace:*", "@reactive-resume/schema": "workspace:*", "@reactive-resume/utils": "workspace:*", + "fflate": "^0.8.3", "zod": "^4.6.5" }, "devDependencies": { diff --git a/packages/import/src/html.ts b/packages/import/src/html.ts index 7459b5f36..c8b3536ed 100644 --- a/packages/import/src/html.ts +++ b/packages/import/src/html.ts @@ -29,3 +29,29 @@ export function arrayToHtmlList(items: string[]): string { if (items.length === 0) return ""; return ``; } + +export const BULLET_PATTERN = /^\s*[-–—•*◦‣·]\s+/; + +const escapeHtml = (value: string) => + value + .replace(/&/g, "&") + .replace(//g, ">") + .replace(/"/g, """) + .replace(/'/g, "'"); + +/** + * Converts plain-text lines into escaped HTML: a