From 6af317ffc23424fef612f3b0382ad718839a1513 Mon Sep 17 00:00:00 2001 From: Codex Date: Sun, 6 Sep 2026 01:54:03 +0900 Subject: [PATCH] fix: audit post emphasis at Korean punctuation boundaries --- ...laude-code-weekly-limits-september-2026.md | 2 +- .../posts/gpt-6-astra-vs-claude-fable-5-1.md | 8 +-- ...korea-ai-third-place-dokpamo-evaluation.md | 8 +-- e2e/markdown-emphasis.spec.ts | 55 +++++++++++++++++++ tests/markdown-cjk-emphasis.test.ts | 17 ++++++ tests/post-currency.test.ts | 10 ++++ tests/post-emphasis.test.ts | 51 +++++++++++++++++ 7 files changed, 142 insertions(+), 9 deletions(-) create mode 100644 e2e/markdown-emphasis.spec.ts create mode 100644 tests/post-currency.test.ts create mode 100644 tests/post-emphasis.test.ts diff --git a/content/posts/claude-code-weekly-limits-september-2026.md b/content/posts/claude-code-weekly-limits-september-2026.md index 6dc4da1..6eb5158 100644 --- a/content/posts/claude-code-weekly-limits-september-2026.md +++ b/content/posts/claude-code-weekly-limits-september-2026.md @@ -42,7 +42,7 @@ Anthropic은 2026년 5월 6일 공식 발표에서 Pro, Max, Team, seat-based En - [Anthropic: Higher usage limits for Claude and a compute deal with SpaceX](https://www.anthropic.com/news/higher-limits-spacex) -이번 9월 변경의 핵심은 이 5시간 한도가 아니라 **주간 한도(weekly limit)**다. +이번 9월 변경의 핵심은 이 5시간 한도가 아니라 **주간 한도**(weekly limit)다. Anthropic 도움말도 사용량 제한이 대화 길이, 복잡성, 사용하는 모델과 기능 등에 영향을 받으며, Claude.ai와 Claude Code 같은 여러 제품 영역의 사용량이 서로 영향을 줄 수 있다고 설명한다. diff --git a/content/posts/gpt-6-astra-vs-claude-fable-5-1.md b/content/posts/gpt-6-astra-vs-claude-fable-5-1.md index cc44762..ca1e10a 100644 --- a/content/posts/gpt-6-astra-vs-claude-fable-5-1.md +++ b/content/posts/gpt-6-astra-vs-claude-fable-5-1.md @@ -28,7 +28,7 @@ GPT-6 Astra에서는 비동기 도구 호출과 작업 도중 사용자 개입 OpenAI는 GPT-6 Astra를 복잡한 end-to-end 작업을 위한 자사의 가장 강력한 모델로 소개한다. -OpenAI가 공개한 비교에서 [AutomationBench](https://openai.com/index/gpt-6-astra/) 점수는 GPT-6 Astra가 **41.4%**, Claude Fable 5.1이 **31.4%**다. +OpenAI가 공개한 비교에서 [AutomationBench](https://openai.com/index/gpt-6-astra/) 점수는 GPT-6 Astra가 **41.4%**, Claude Fable 5.1이 **31.4**%다. AutomationBench는 여러 단계의 실제 업무 자동화 능력을 평가하는 벤치마크라서 이번 글의 주제인 에이전트 작업과도 꽤 직접적으로 연결된다. @@ -83,7 +83,7 @@ GPT-6 Astra의 모델 사양은 눈에 띈다. [공식 모델 문서](https://developers.openai.com/api/docs/models/gpt-6-astra)에 따르면 Astra는 **1,050,000 토큰 컨텍스트 윈도우**와 **128,000 토큰 최대 출력**을 지원한다. -가격은 100만 토큰 기준 입력 **$10**, 캐시 입력 **$1**, 출력 **$50**이다. 272K 입력 토큰을 넘는 요청은 전체 요청에 더 높은 장문 요금이 적용된다. +가격은 100만 토큰 기준 입력 **\$10**, 캐시 입력 **\$1**, 출력 **\$50**이다. 272K 입력 토큰을 넘는 요청은 전체 요청에 더 높은 장문 요금이 적용된다. 하지만 더 중요한 변화는 이 숫자들이 아니다. @@ -135,7 +135,7 @@ OpenAI가 [Responses API를 신규 프로젝트에 권장](https://developers.op Anthropic의 접근에서 특히 눈에 들어오는 것은 cache read 가격이다. -Claude Fable 5.1의 기본 가격은 100만 토큰 기준 입력 **$10**, 출력 **$50**으로 Fable 5와 같다. +Claude Fable 5.1의 기본 가격은 100만 토큰 기준 입력 **\$10**, 출력 **\$50**으로 Fable 5와 같다. 그런데 Anthropic은 [cache read 가격을 75% 낮춰 100만 토큰당 $0.25](https://www.anthropic.com/claude-fable-and-mythos-5-1)로 변경했다. @@ -168,7 +168,7 @@ Fable 5.1은 반복 컨텍스트가 많은 작업에서 **캐시 비용을 크 그리고 모델마다 하나의 작업을 끝내기 위해 소비하는 토큰과 도구 호출 횟수도 다를 수 있다. -결국 개발자가 봐야 하는 숫자는 점점 `cost per token`보다 **`cost per completed task`**에 가까워진다. +결국 개발자가 봐야 하는 숫자는 점점 `cost per token`보다 **`cost per completed task`에** 가까워진다. 비싼 모델이 더 적은 시도와 토큰으로 작업을 끝내면 결과적으로 더 저렴할 수도 있다. 반대로 토큰 단가는 같아도 긴 컨텍스트를 계속 다시 읽거나 실패와 재시도가 많으면 실제 비용은 올라간다. diff --git a/content/posts/korea-ai-third-place-dokpamo-evaluation.md b/content/posts/korea-ai-third-place-dokpamo-evaluation.md index cb0e72b..836b5f8 100644 --- a/content/posts/korea-ai-third-place-dokpamo-evaluation.md +++ b/content/posts/korea-ai-third-place-dokpamo-evaluation.md @@ -22,7 +22,7 @@ draft: false Artificial Analysis 기준 전체 모델 가운데 10위권에 해당하는 성적이었고, 미국과 중국을 제외하면 특히 눈에 띄는 결과였다. -Artificial Analysis는 이후 한국을 미국과 중국에 이어 **"clear #3 nation in AI"**라고 평가했다. +Artificial Analysis는 이후 한국을 미국과 중국에 이어 "**clear #3 nation in AI**"라고 평가했다. 그런데 조금 이상한 일이 벌어졌다. @@ -46,7 +46,7 @@ Artificial Analysis는 이후 한국을 미국과 중국에 이어 **"clear #3 n 정부는 GPU, 데이터, 인프라 등을 지원하고 여러 정예팀이 경쟁하는 방식으로 프로젝트를 진행하고 있다. -여기서 중요한 단어는 **'파운데이션 모델'**이다. +여기서 중요한 단어는 '**파운데이션 모델**'이다. ChatGPT와 비슷한 챗봇 서비스를 하나 만드는 프로젝트가 아니다. @@ -331,7 +331,7 @@ Web Interface ## 그렇다면 왜 '세계 3위'를 이야기할 때는 AAII를 사용했을까 -Artificial Analysis는 최근 한국을 미국과 중국에 이어 **"clear #3 nation in AI"**라고 평가했다. +Artificial Analysis는 최근 한국을 미국과 중국에 이어 "**clear #3 nation in AI**"라고 평가했다. 중요한 것은 하나의 모델만 보고 내린 평가가 아니라는 점이다. @@ -343,7 +343,7 @@ Artificial Analysis는 최근 한국을 미국과 중국에 이어 **"clear #3 n 한국의 글로벌 AI 경쟁력을 설명할 때는 AAII라는 글로벌 모델 성능 지표가 강력한 근거가 된다. -그런데 독파모에서 지원팀을 선정할 때 같은 지표의 비중은 **25%**다. +그런데 독파모에서 지원팀을 선정할 때 같은 지표의 비중은 **25**%다. 즉, diff --git a/e2e/markdown-emphasis.spec.ts b/e2e/markdown-emphasis.spec.ts new file mode 100644 index 0000000..f616953 --- /dev/null +++ b/e2e/markdown-emphasis.spec.ts @@ -0,0 +1,55 @@ +import { expect, test } from "@playwright/test" + +const cases = [ + { + slug: "korea-ai-third-place-dokpamo-evaluation", + phrases: [ + ["clear #3 nation in AI", 2], + ["파운데이션 모델", 1], + ["25", 1], + ], + }, + { slug: "claude-code-weekly-limits-september-2026", phrases: [["주간 한도", 1]] }, + { + slug: "gpt-6-astra-vs-claude-fable-5-1", + phrases: [ + ["완벽하지는 않지만 빨랐다", 1], + ["31.4", 1], + ["cost per completed task에", 1], + ["$10", 2], + ["$1", 1], + ["$50", 2], + ], + }, +] as const + +for (const { slug, phrases } of cases) { + test(`${slug} displays emphasis on the built page`, async ({ page }, testInfo) => { + await page.goto(`/posts/${slug}/`) + const article = page.locator("article[data-pagefind-body]") + await expect(article).toBeVisible() + for (const [phrase, count] of phrases) { + const strong = article + .locator("strong") + .filter({ hasText: new RegExp(`^${phrase.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}$`) }) + await expect(strong).toHaveCount(count) + for (let index = 0; index < count; index++) { + const item = strong.nth(index) + await item.scrollIntoViewIfNeeded() + await expect(item).toBeVisible() + expect( + await item.evaluate((element) => + Number.parseInt(getComputedStyle(element).fontWeight, 10), + ), + ).toBeGreaterThanOrEqual(600) + const paragraph = item.locator("xpath=ancestor::p[1]") + await expect(paragraph).not.toContainText("**") + await expect(paragraph.locator(".katex")).toHaveCount(0) + await testInfo.attach(`${phrase}-${index}`, { + body: await paragraph.screenshot(), + contentType: "image/png", + }) + } + } + }) +} diff --git a/tests/markdown-cjk-emphasis.test.ts b/tests/markdown-cjk-emphasis.test.ts index 29678dd..f9e8ce8 100644 --- a/tests/markdown-cjk-emphasis.test.ts +++ b/tests/markdown-cjk-emphasis.test.ts @@ -11,3 +11,20 @@ describe("markdown emphasis around Korean particles", () => { expect(html).not.toContain("**완벽하지는 않지만 빨랐다**") }) }) + +it.each([ + ['"**인용**"이라고', '"인용"이라고'], + ["'**용어**'이다", "'용어'이다"], + ["**한도**(limit)다", "한도(limit)다"], + ["**25**%다", "25%다"], + ["**`task`에** 가깝다", "task에 가깝다"], + [ + "**\\$10**, **\\$1**, **\\$50**이다", + "$10, $1, $50이다", + ], + ["**문장.**\n\n**평가 기준**이다", "평가 기준이다"], +])("renders safe punctuation boundaries: %s", async (source, expected) => { + const { html } = await renderMarkdown(source) + expect(html).toContain(expected) + expect(html).not.toContain("**") +}) diff --git a/tests/post-currency.test.ts b/tests/post-currency.test.ts new file mode 100644 index 0000000..4be83a0 --- /dev/null +++ b/tests/post-currency.test.ts @@ -0,0 +1,10 @@ +import { expect, it } from "vitest" +import { renderMarkdown } from "../src/lib/markdown" +import { getPostBySlug } from "../src/lib/posts" + +it("currency remains strong prose", async () => { + const { html } = await renderMarkdown(getPostBySlug("gpt-6-astra-vs-claude-fable-5-1").content) + expect(html.match(/\$10<\/strong>/g)).toHaveLength(2) + expect(html.match(/\$50<\/strong>/g)).toHaveLength(2) + expect(html).toContain("$1") +}) diff --git a/tests/post-emphasis.test.ts b/tests/post-emphasis.test.ts new file mode 100644 index 0000000..669b09f --- /dev/null +++ b/tests/post-emphasis.test.ts @@ -0,0 +1,51 @@ +import { readdirSync, readFileSync } from "node:fs" +import { join } from "node:path" +import { Window } from "happy-dom" +import { visit } from "unist-util-visit" +import { describe, expect, it } from "vitest" +import { renderMarkdown } from "../src/lib/markdown" +import { createMarkdownProcessor } from "../src/lib/markdown/render" +import { getPostBySlug } from "../src/lib/posts" + +const directory = join(process.cwd(), "content/posts") + +describe("all authored posts preserve emphasis", () => { + for (const file of readdirSync(directory).filter((name) => name.endsWith(".md"))) { + it(`${file} has no unparsed emphasis delimiters in prose`, async () => { + const source = readFileSync(join(directory, file), "utf8") + const body = source.replace(/^---\r?\n[\s\S]*?\r?\n---\r?\n/, "") + const tree = createMarkdownProcessor().parse(body) + const failures: string[] = [] + visit(tree, "text", (node) => { + if (/\*\*|__/.test(node.value)) { + failures.push(`line ${node.position?.start.line}: ${node.value}`) + } + }) + expect(failures).toEqual([]) + const { html } = await renderMarkdown(body) + const window = new Window() + try { + window.document.body.innerHTML = html + for (const code of window.document.querySelectorAll("pre, code")) code.remove() + expect(window.document.body.textContent).not.toMatch(/\*\*|__/) + } finally { + await window.happyDOM.close() + } + }) + } +}) + +describe("real post rendering regressions", () => { + it.each([ + ["korea-ai-third-place-dokpamo-evaluation", "clear #3 nation in AI", 2], + ["korea-ai-third-place-dokpamo-evaluation", "파운데이션 모델", 1], + ["korea-ai-third-place-dokpamo-evaluation", "25", 1], + ["claude-code-weekly-limits-september-2026", "주간 한도", 1], + ["gpt-6-astra-vs-claude-fable-5-1", "31.4", 1], + ["gpt-6-astra-vs-claude-fable-5-1", "완벽하지는 않지만 빨랐다", 1], + ["gpt-6-astra-vs-claude-fable-5-1", "cost per completed task에", 1], + ])("%s renders %s as strong", async (slug, text, count) => { + const { html } = await renderMarkdown(getPostBySlug(slug).content) + expect(html.split(`${text}`).length - 1).toBe(count) + }) +})