From d6168dc59ba25132a463132b4383f7f3c98d137a Mon Sep 17 00:00:00 2001 From: Sameer Khan <101021315+sameerkhansf@users.noreply.github.com> Date: Tue, 8 Sep 2026 11:59:58 -0700 Subject: [PATCH] fix(content): correct two citation/spec errors; fix the rule that missed one MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three live posts corrected, and the guardrail I added in #86 replaced because it would not have caught the case that prompted it. 1. minicpm5-2b-review-2026: cited arXiv:2506.07900 as MiniCPM5-2B's technical report. That paper is "MiniCPM4: Ultra-Efficient LLMs on End Devices" (June 2025) — the previous generation. The card's other tag, 2602.09003, is "Data Science and Technology Towards AGI Part I: Tiered Data Management". Neither describes MiniCPM5, so the post now says so instead of promoting a near-miss. Both titles read off arxiv.org/abs/. 2. glm-5-3-review-2026 and glm-5-3-flash-vs-qwen3-8-flash-next-comparison-2026: both listed the context length as 300K. That number is the evaluation context for the HLE benchmark in the cards' footnotes; config.json gives max_position_embeddings = 1048576 for both models. Same failure as the pricing bug — a footnote figure promoted to a spec. 3. The #86 rule said "prefer the newest arxiv tag unless the card ties an older one to this version". Applied to MiniCPM5-2B that picks 2602.09003, which is also wrong. Any ordering heuristic fails here, because the tag list carries no guarantee at all. Replaced with: fetch arxiv.org/abs/, read the title, and cite it only if the title names this model and version. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01Vru5xPN55JPU7dyVS3uswd --- .github/workflows/daily-til.lock.yml | 2 +- .github/workflows/daily-til.md | 2 +- ...m-5-3-flash-vs-qwen3-8-flash-next-comparison-2026.mdx | 4 ++-- content/blog/glm-5-3-review-2026.mdx | 9 +++++++-- content/blog/minicpm5-2b-review-2026.mdx | 9 +++++++-- 5 files changed, 18 insertions(+), 8 deletions(-) diff --git a/.github/workflows/daily-til.lock.yml b/.github/workflows/daily-til.lock.yml index 2da3094..185483d 100644 --- a/.github/workflows/daily-til.lock.yml +++ b/.github/workflows/daily-til.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"2d45b25323bfefe5fb308930109b11eeb8f932e2108d6dca35ef983bdb3d446f","body_hash":"d92d9cb789d71ffbf4a7c2ba14029095b93a186c42eb8c0e27954f7fd0785869","compiler_version":"v0.87.10","strict":true,"engine_base_url_customized":true,"agent_id":"copilot","detection_agent_id":"copilot","engine_versions":{"copilot":"1.0.80"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"2d45b25323bfefe5fb308930109b11eeb8f932e2108d6dca35ef983bdb3d446f","body_hash":"d8367bb395bfbab60242ea2d0bc0ad0f5ddbf192ef9536a5e409374a33c805b5","compiler_version":"v0.87.10","strict":true,"engine_base_url_customized":true,"agent_id":"copilot","detection_agent_id":"copilot","engine_versions":{"copilot":"1.0.80"}} # gh-aw-manifest: {"version":1,"secrets":["COPILOT_GITHUB_TOKEN","GH_AW_CI_TRIGGER_TOKEN","GH_AW_DEFAULT_OTLP_HEADERS","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN","OPENROUTER_API_KEY"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"bc8c008a419c5b7a29df6f5641edd35fd1c6ea85","version":"v0.87.10"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.10","digest":"sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.10@sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.10","digest":"sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.10@sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.10","digest":"sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.10@sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.14","digest":"sha256:b2f0c2b2f17b5fbe809e5bb99dc185b6ddd70df25295dc63a6d526350334eff5","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.14@sha256:b2f0c2b2f17b5fbe809e5bb99dc185b6ddd70df25295dc63a6d526350334eff5"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:bac2192f6374d6262116399b34fc5e143d576f82719e90a18261cae7480f4d4e","pinned_image":"ghcr.io/github/gh-aw-node@sha256:bac2192f6374d6262116399b34fc5e143d576f82719e90a18261cae7480f4d4e"},{"image":"ghcr.io/github/github-mcp-server:v1.11.0","digest":"sha256:fbec75de11c255213fa08d80fb166abe73d851fff631c51c0079872967720699","pinned_image":"ghcr.io/github/github-mcp-server:v1.11.0@sha256:fbec75de11c255213fa08d80fb166abe73d851fff631c51c0079872967720699"}],"mcp_servers":[{"name":"github","tools":["get_commit","get_file_contents","get_latest_release","get_me","get_pull_request","get_pull_request_comments","get_pull_request_diff","get_pull_request_files","get_pull_request_review_comments","get_pull_request_reviews","get_pull_request_status","get_release_by_tag","get_tag","issue_read","list_branches","list_commits","list_issue_types","list_issues","list_pull_requests","list_releases","list_starred_repositories","list_tags","pull_request_read","search_code","search_issues","search_pull_requests","search_repositories"]},{"name":"safeoutputs","tools":["create_pull_request","missing_data","missing_tool","noop"]}]} # This file was automatically generated by gh-aw (v0.87.10). DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/daily-til.md b/.github/workflows/daily-til.md index be3096b..32bda63 100644 --- a/.github/workflows/daily-til.md +++ b/.github/workflows/daily-til.md @@ -142,7 +142,7 @@ Read all five with a single `cat` each, then choose the topic. Never run shell s - **Description**: one-sentence summary of the verdict/scope, 40-320 chars. - **Body**: markdown tables for comparisons (the corpus uses them heavily) — every table cell padded with one space on each side of every pipe, like `| Model | Price |` (compact `|Model|Price|` fails lint); fenced code blocks with a language wherever commands or config appear; citations ONLY as `[label](https://...)` — never `[[url]]` wiki-links and never bare URLs, including in source tables; a blank line before and after every heading and every list; file ends with a newline; a `<` followed by a letter or digit in prose (`<50ms`, ``) is JSX to MDX and fails the compile (so is a bare `{`) — write `under 50ms`, escape it as `\<50ms` / `\{`, or put it in backticks. - **Sources must be reachable**: only domains on this workflow's network allowlist can be fetched (github.com, huggingface.co, openai.com, anthropic.com, ai.google.dev, blog.google, mistral.ai, deepseek.com, z.ai, qwencloud.com, python.org and their subdomains). Prefer candidates whose primary sources live there; if a candidate's key sources are blocked by the firewall, pick a different candidate rather than writing from memory. - - **Cite the paper that describes THIS release.** A model card often carries several `arxiv:` tags — the current technical report plus its predecessors. Picking the first or the oldest misattributes the work: PR #77 cited arXiv 2310.10688 (the 2023 original TimesFM paper) as TimesFM 3.0's reference, and PR #85 cited arXiv 2506.07900 as MiniCPM5-2B's technical report while the card also lists the newer 2602.09003. Read every `arxiv:` tag, prefer the newest one unless the card explicitly ties an older ID to this version, and if you cannot tell which describes this release, cite the model card and say the technical report was not identified. + - **Open every paper you cite and read its title before citing it.** An `arxiv:` tag on a model card is not a promise that the paper describes that model — cards routinely tag a predecessor's report, or an unrelated lab paper. Never infer a paper's subject from its ID, its position in the tag list, or its date. Fetch `https://arxiv.org/abs/` and read the actual title: if it does not name this model and version, it is not this model's technical report. Say "no technical report is linked from the model card" rather than promote the closest-looking tag. Both previous failures came from guessing: PR #77 cited arXiv 2310.10688 as TimesFM 3.0's reference when it is the 2023 original TimesFM paper, and PR #85 cited arXiv 2506.07900 as MiniCPM5-2B's technical report when its title is "MiniCPM4: Ultra-Efficient LLMs on End Devices" — that card's other tag, 2602.09003, is a data-management paper, so neither was a match and picking "the newer one" would also have been wrong. - **Prices come from the vendor's price sheet, never from arithmetic.** A relative claim ("one-tenth the price", "half the cost") is worthless without its baseline: read the sentence and name what it is cheaper *than* — vendors almost always mean their own previous model, not a competitor. Then fetch the vendor's pricing page (`docs.z.ai/guides/overview/pricing`, `qwencloud.com/models/`, `openai.com/api/pricing`, `anthropic.com/pricing`) and quote the listed per-token numbers with that link. If no price is published, the cell reads "no official listing" — never multiply or divide some other model's price to produce a dollar figure, and never present a derived number as a price. - **Never**: fabricated testing claims, "In today's fast-paced world" intros, unsupported superlatives, uncited numbers, emoji anywhere (headings, tables, lists — use "Yes"/"No" in comparison tables, plain words everywhere else). - **Every contender must be real and fetched**: each row of a comparison table names a specific product you fetched a primary source for this run, with that source linked in the row or the section about it. Never invent placeholder contenders ("Generic X Server", "X Toolkit") to fill a table. If you cannot source at least three real contenders, write a single-product review of the one you can source instead of a comparison. diff --git a/content/blog/glm-5-3-flash-vs-qwen3-8-flash-next-comparison-2026.mdx b/content/blog/glm-5-3-flash-vs-qwen3-8-flash-next-comparison-2026.mdx index 9e0f5e5..ee8d2b5 100644 --- a/content/blog/glm-5-3-flash-vs-qwen3-8-flash-next-comparison-2026.mdx +++ b/content/blog/glm-5-3-flash-vs-qwen3-8-flash-next-comparison-2026.mdx @@ -19,7 +19,7 @@ GLM-5.3-Flash and Qwen3.8-Flash-Next represent two approaches to efficient large | Model | Total Parameters | Active Parameters | Context Length | Price (per M tokens) | Best For | | ------- | ------------------ | ------------------- | ---------------- | ---------------------- | ---------- | -| GLM-5.3-Flash | 320B | 18B | 300K tokens | $0.15 in / $0.50 out* | Multimodal agentic workflows | +| GLM-5.3-Flash | 320B | 18B | 1M tokens | $0.15 in / $0.50 out* | Multimodal agentic workflows | | Qwen3.8-Flash-Next | 125B | 6B + 51B n-gram | 262K tokens (1M with YaRN) | No official listing† | Long-horizon reasoning & tool use | \*List price from the [Z.ai pricing page](https://docs.z.ai/guides/overview/pricing) (cached input $0.03). A 50% launch discount ran through 2026-09-09. @@ -144,7 +144,7 @@ Both models recommend specific settings for different modes: ### Context Management -- GLM-5.3-Flash uses explicit context management strategies for its 300K token evaluations +- GLM-5.3-Flash's window is 1M tokens (`max_position_embeddings` = 1,048,576); the 300K figure in its benchmark footnotes is the evaluation context for HLE, not the model limit - Qwen3.8-Flash-Next natively supports 262K tokens and recommends YaRN for extension beyond that - Both require careful output length allocation for agentic workflows (separate reasoning vs. final response limits) diff --git a/content/blog/glm-5-3-review-2026.mdx b/content/blog/glm-5-3-review-2026.mdx index 23b8db0..1b694e9 100644 --- a/content/blog/glm-5-3-review-2026.mdx +++ b/content/blog/glm-5-3-review-2026.mdx @@ -2,6 +2,7 @@ title: "GLM-5.3 Review: Z.ai's New Open Weights Model for Coding and Agentic Workflows" description: "Comprehensive review of GLM-5.3, Z.ai's latest open weights Mixture-of-Experts model showing strong performance on coding benchmarks and cybersecurity tasks." date: "2026-09-03" +updated: "2026-09-08" author: "Sameer Khan" tags: ["AI", "GLM", "Z.ai", "LLM", "Developer Tools", "Agentic AI"] category: "AI" @@ -29,7 +30,7 @@ GLM-5.3 was released by Z.ai in August 2026 as the latest iteration in their GLM | AutomationBench (v1.0.6) | **48.2** | 26.2 | 43.2 | 39.8 | 41.0 | 45.8 | | HLE w/ Tools | 62.5 | 54.7 | 60.0 | 56.2 | 57.9 | **64.5** | -**Context Length:** 300K tokens (with context management strategy) +**Context Length:** 1M tokens (`max_position_embeddings` = 1,048,576 in [config.json](https://huggingface.co/zai-org/GLM-5.3/blob/main/config.json)) **Reasoning Effort:** Configurable (low, high, max) - defaults to max **License:** Other (GLM-5.3 license) **Deployment:** Compatible with SGLang, vLLM, TokenSpeed, Transformers, KTransformers, Unsloth, and Ascend NPU platforms @@ -102,7 +103,7 @@ As an open weights model, GLM-5.3 can be run locally at no licensing cost (beyon ### Limitations - The GLM-5.3 license contains restrictions that may affect certain use cases -- Context length of 300K requires careful management for optimal performance +- Long-context runs need careful management: Z.ai's own evaluations cap context per benchmark (300K for HLE, 400K for DeepSWE and Terminal-Bench 3.0, 1M for NL2Repo and ALE) rather than using the full window - Some benchmarks show trailing performance compared to GPT-5.6 Sol - Limited public API availability information in primary sources @@ -135,3 +136,7 @@ For developers focused on coding agents, automation workflows, or cybersecurity - GLM-5.3 model card and README: [GLM-5.3 model card and README](https://huggingface.co/zai-org/GLM-5.3) (accessed 2026-09-03) - GLM-5.3 API metadata: [GLM-5.3 API metadata](https://huggingface.co/api/models/zai-org/GLM-5.3) (accessed 2026-09-03) + +--- + +*Correction (2026-09-08): the original version listed the context length as 300K tokens. That figure is the evaluation context used for the HLE benchmark in the model card's footnotes, not the model's window; `config.json` gives `max_position_embeddings` = 1,048,576.* diff --git a/content/blog/minicpm5-2b-review-2026.mdx b/content/blog/minicpm5-2b-review-2026.mdx index 5f09973..09cf3f3 100644 --- a/content/blog/minicpm5-2b-review-2026.mdx +++ b/content/blog/minicpm5-2b-review-2026.mdx @@ -2,6 +2,7 @@ title: "MiniCPM5-2B Review: 2B Parameter Model Excels in Coding and Agent Tasks" description: "Review of OpenBMB's MiniCPM5-2B, a dense 2B Transformer model achieving 2B-class SOTA performance with strong capabilities in coding, mathematics, and tool use." date: "2026-09-08" +updated: "2026-09-08" author: "Sameer Khan" tags: ["AI", "LLM", "MiniCPM", "OpenBMB", "2B model"] category: "AI" @@ -70,7 +71,7 @@ MiniCPM5-2B is readily available through: - Hugging Face Model Hub: `openbmb/MiniCPM5-2B` - GitHub Repository: [OpenBMB/MiniCPM](https://github.com/OpenBMB/MiniCPM) - Online Demo: Hugging Face Spaces -- Technical Report: arXiv:2506.07900 +- Predecessor technical report: [MiniCPM4: Ultra-Efficient LLMs on End Devices](https://arxiv.org/abs/2506.07900) (arXiv:2506.07900, June 2025) — covers MiniCPM4, not this release; no MiniCPM5 technical report is linked from the model card The model is part of the broader MiniCPM ecosystem, which includes various sizes and specialized variants to meet different deployment needs. @@ -86,4 +87,8 @@ For teams working on resource-constrained applications or specialized developer [2] MiniCPM5-2B README. Hugging Face. Accessed 2026-09-08. [https://huggingface.co/openbmb/MiniCPM5-2B/resolve/main/README.md](https://huggingface.co/openbmb/MiniCPM5-2B/resolve/main/README.md) -[3] MiniCPM Technical Report. arXiv. Accessed 2026-09-08. [https://arxiv.org/pdf/2506.07900](https://arxiv.org/pdf/2506.07900) +[3] MiniCPM4: Ultra-Efficient LLMs on End Devices. arXiv, June 2025. Accessed 2026-09-08. [https://arxiv.org/abs/2506.07900](https://arxiv.org/abs/2506.07900) + +--- + +*Correction (2026-09-08): the original version cited arXiv:2506.07900 as MiniCPM5-2B's technical report. That paper is "MiniCPM4: Ultra-Efficient LLMs on End Devices" (June 2025) and describes the previous generation. The model card carries a second tag, arXiv:2602.09003, which is an unrelated data-management paper — neither is a MiniCPM5 report, and the card links none.*