From 0d5e9d2bc9ecd417b8337ded0f293db6f1a0d73a Mon Sep 17 00:00:00 2001 From: lcy-seso Date: Thu, 1 Oct 2026 21:36:03 +0800 Subject: [PATCH 1/2] [Docs] Add a Blog section, restate what TileOPs is for on Home, and style links in prose The Blog section holds technical explorations from building TileOPs. It starts with its index page only; posts are added to nav and to the index as they are published. The section is built from plain pages rather than Material's blog plugin: under mkdocs-static-i18n that plugin renders no posts and warns on its archive pages. hooks.py gives a post's subtitle its class, so no Markdown page carries styling. Home's introduction now states what TileOPs is for: a library designed for agents and built by them, held to code-quality requirements so the project stays maintainable, its results verifiable and its kernels tunable. Every link in running text carries an arrow, east within the site and north-east off it, and a teal wash on hover. --- CLAUDE.md | 25 ++++++++++++++---- docs/assets/extra.css | 60 +++++++++++++++++++++++++++++++++++++++++++ docs/blog/index.md | 3 +++ docs/blog/index.zh.md | 3 +++ docs/index.md | 39 ++++++++++++++-------------- docs/index.zh.md | 14 ++++------ hooks.py | 14 ++++++++++ mkdocs.yml | 3 +++ 8 files changed, 127 insertions(+), 34 deletions(-) create mode 100644 docs/blog/index.md create mode 100644 docs/blog/index.zh.md diff --git a/CLAUDE.md b/CLAUDE.md index 1be1a475..7d541d3a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -78,9 +78,10 @@ implementation of the same op on that workload. English at the site root, Chinese under `/zh/`. A Chinese page is a `.zh.md` beside the English `.md`, full prose, never an `include-markdown` shell. `backends.md`, `torch-compile.md`, everything under -`performance-guides/memory-bound/` and the two guides under `user-guide/manifest/` -and `user-guide/dispatch/` were authored in Chinese: edit the `.zh.md` first, then -bring the English page in line. Everything else goes the other way. +`performance-guides/memory-bound/` and `blog/`, and the two guides under +`user-guide/manifest/` and `user-guide/dispatch/` were authored in Chinese: edit +the `.zh.md` first, then bring the English page in line. Everything else goes +the other way. | Rule | Detail | |------|--------| @@ -99,8 +100,8 @@ bring the English page in line. Everything else goes the other way. ## Nav -`nav` in `mkdocs.yml` is the page list; its six sections run in the reader's -order, Design last as contributor-facing. +`nav` in `mkdocs.yml` is the page list: Home, then Blog, then the remaining +sections in the reader's order, Design last as contributor-facing. - Add a new page to `nav`, and its label to `nav_translations`. - Put a user-facing topic under User Guide. @@ -111,11 +112,25 @@ order, Design last as contributor-facing. - Leave `toc.integrate` off: the page TOC renders in the right column, and it is incompatible with `navigation.indexes`. +## Blog + +`docs/blog/`: plain pages, not Material's `blog` plugin. Why: the plugin +renders no post under `mkdocs-static-i18n` and warns on its archive pages. + +- One post is `blog/.zh.md` and `blog/.md`, listed in `nav` under + Blog and as one link on `blog/index.md`, newest first. +- A post opens with a short H1 and a one-line subtitle paragraph. `hooks.py` + gives that paragraph the `post-subtitle` class that `extra.css` styles; keep + classes and attribute lists out of the post's Markdown. +- No date line, no in-page TOC: the right column carries the TOC. + ## Conventions - Measure every number and state its conditions. Say when a count will drift. - Admonitions (`!!! note`, `!!! warning`) for callouts; relative Markdown links for internal cross-references. +- Write links as plain Markdown. `extra.css` gives every link in running text an + arrow, east within the site and north-east off it, and a teal wash on hover. - Link to the TileOPs repo rather than duplicating it. A page authored here that mirrors upstream content will drift. - Gitignored: `site/`, `__pycache__/`, `.cache/`, `TileOPs/`, and diff --git a/docs/assets/extra.css b/docs/assets/extra.css index b323dcd1..a27779fa 100644 --- a/docs/assets/extra.css +++ b/docs/assets/extra.css @@ -438,6 +438,54 @@ body, color: var(--tf-teal); } +/* Every link in running text carries an arrow, so it reads as a link before it + * is hovered: east for a page on this site, north-east for one that leaves it. + * Hover lifts a teal wash from the underline and nudges the arrow. Scoped to + * prose — paragraphs, list items, table cells — so headings' permalinks, the + * nav, buttons, badges and the Home card keep their own look, and no page's + * Markdown has to mark a link for it. + */ +.md-typeset :is(p, li, td, dd) a:not(:has(img)) { + padding: 0 0.08em; + margin: 0 -0.08em; + border-radius: 0.15em; + background: linear-gradient(var(--tf-teal-tint), var(--tf-teal-tint)) 0 100% / 100% 0 no-repeat; + transition: background-size 160ms ease, text-decoration-color 160ms ease; +} + +.md-typeset :is(p, li, td, dd) a:not(:has(img)):is(:hover, :focus-visible) { + background-size: 100% 100%; +} + +.md-typeset :is(p, li, td, dd) a:not(:has(img))::after { + content: ""; + display: inline-block; + width: 0.62em; + height: 0.62em; + margin-left: 0.18em; + vertical-align: 0.06em; + background-color: currentcolor; + -webkit-mask: url("data:image/svg+xml,%3Csvg%20xmlns='http://www.w3.org/2000/svg'%20viewBox='0%200%2012%2012'%3E%3Cpath%20d='M2%206h8M6.5%202.5%2010%206l-3.5%203.5'%20fill='none'%20stroke='%23000'%20stroke-width='1.6'%20stroke-linecap='round'%20stroke-linejoin='round'/%3E%3C/svg%3E") center / contain no-repeat; + mask: url("data:image/svg+xml,%3Csvg%20xmlns='http://www.w3.org/2000/svg'%20viewBox='0%200%2012%2012'%3E%3Cpath%20d='M2%206h8M6.5%202.5%2010%206l-3.5%203.5'%20fill='none'%20stroke='%23000'%20stroke-width='1.6'%20stroke-linecap='round'%20stroke-linejoin='round'/%3E%3C/svg%3E") center / contain no-repeat; + opacity: 0.7; + transition: transform 160ms ease, opacity 160ms ease; +} + +.md-typeset :is(p, li, td, dd) a[href^="http"]:not(:has(img))::after { + vertical-align: 0.12em; + -webkit-mask-image: url("data:image/svg+xml,%3Csvg%20xmlns='http://www.w3.org/2000/svg'%20viewBox='0%200%2012%2012'%3E%3Cpath%20d='M4%202h6v6M10%202%202.5%209.5'%20fill='none'%20stroke='%23000'%20stroke-width='1.6'%20stroke-linecap='round'%20stroke-linejoin='round'/%3E%3C/svg%3E"); + mask-image: url("data:image/svg+xml,%3Csvg%20xmlns='http://www.w3.org/2000/svg'%20viewBox='0%200%2012%2012'%3E%3Cpath%20d='M4%202h6v6M10%202%202.5%209.5'%20fill='none'%20stroke='%23000'%20stroke-width='1.6'%20stroke-linecap='round'%20stroke-linejoin='round'/%3E%3C/svg%3E"); +} + +.md-typeset :is(p, li, td, dd) a:not(:has(img)):is(:hover, :focus-visible)::after { + opacity: 1; + transform: translateX(0.1em); +} + +.md-typeset :is(p, li, td, dd) a[href^="http"]:not(:has(img)):is(:hover, :focus-visible)::after { + transform: translate(0.08em, -0.08em); +} + /* A rule between sections, drawn as light rather than as a line. */ .md-typeset hr { height: 1px; @@ -1670,3 +1718,15 @@ html[lang="zh"] .md-nav__title { line-height: 1.7; text-align: left; } + +/* A blog post's subtitle: the line under the H1 that says what the post is + * about, so the H1 itself stays one short line. + */ +.md-typeset .post-subtitle { + margin: -0.6rem 0 1.6rem; + color: var(--tf-muted); + font-family: var(--tf-display); + font-size: 1.15rem; + font-weight: 600; + line-height: 1.35; +} diff --git a/docs/blog/index.md b/docs/blog/index.md new file mode 100644 index 00000000..f735cf6c --- /dev/null +++ b/docs/blog/index.md @@ -0,0 +1,3 @@ +# Blog + +Technical explorations from building TileOPs. diff --git a/docs/blog/index.zh.md b/docs/blog/index.zh.md new file mode 100644 index 00000000..428dadf7 --- /dev/null +++ b/docs/blog/index.zh.md @@ -0,0 +1,3 @@ +# 博客 + +TileOPs 开发中的技术探索。 diff --git a/docs/index.md b/docs/index.md index 57f089a4..74686c85 100644 --- a/docs/index.md +++ b/docs/index.md @@ -1,25 +1,23 @@ # TileOPs -TileOPs is an operator library for large-model inference, built on -[TileLang](https://github.com/tile-ai/tilelang). One set of op interfaces can be -implemented by different backends on different hardware. - -TileOPs differs from a hand-written operator library in how it is organised: every -op is first declared as a spec, and an agent then generates the implementation from -that spec. The spec is the only input to code generation and the standard the result -is accepted against: - -- correctness is judged against the reference implementation the spec names; -- performance is judged against the bound the roofline model gives. - -Neither check depends on human judgement. An implementation can therefore be -regenerated from its spec at any time, while a spec cannot be derived from an -implementation. - -To a caller, TileOPs is a set of ops that can be called directly. Shapes and dtype -are fixed at call time; the specialized kernel is built and cached on first use and -can then be used with CUDA graphs. Each op declares whether it supports -`torch.compile(fullgraph=True)`. +TileOPs is an exploratory operator library for large-model inference, built on +[TileLang](https://github.com/tile-ai/tilelang). TileOPs is designed for agents, +and agents build the whole project. Such a project has to meet concrete +code-quality requirements: its structure stays consistent, it does not diverge or +bloat as ops are added, and its code stays maintainable. The design of TileOPs +serves three goals: + +- **Maintainable.** Each op is declared by a spec, and an agent generates the + implementation from it. The ops in a family share one set of interfaces and + rules, and every new op and kernel follows them. +- **Verifiable.** The spec names the reference implementation correctness is + judged against. Tests compare a kernel's output with that reference, and + performance measurements compare its measured speed with the bound the roofline + model gives. The acceptance criteria are fixed in advance, and the checks run + automatically. +- **Tunable.** The roofline model gives the gap between each kernel and its + performance bound. The [nightly benchmarks](benchmarks/index.md) compare each + kernel with the fastest other implementation on the same hardware. ## Installation @@ -46,6 +44,7 @@ flops, nbytes = op.eval_roofline() # what the call had to do and move ## Where to go next +- [Blog](blog/index.md): technical explorations from building TileOPs. - [User Guide](user-guide/index.md): reading and writing the manifest, bringing an op into `torch.compile`, how a benchmark is timed, and adding a hardware backend. - [API Reference](api/index.md): the constructor parameters and call signatures of diff --git a/docs/index.zh.md b/docs/index.zh.md index 1eaf4269..d3d770d6 100644 --- a/docs/index.zh.md +++ b/docs/index.zh.md @@ -1,15 +1,10 @@ # TileOPs -TileOPs 是一个面向大模型推理的算子库,构建在 [TileLang](https://github.com/tile-ai/tilelang) 之上。同一套 op 接口可以由不同 backend 在不同硬件上实现。 +TileOPs 是一个面向大模型推理的探索性算子库,构建在 [TileLang](https://github.com/tile-ai/tilelang) 之上。TileOPs 为 agent 设计,整个项目由 agent 构建。这样的项目需要满足明确的代码质量要求:项目结构保持一致,不随 op 增多而发散或膨胀,代码可以长期维护。TileOPs 的设计围绕三个目标: -TileOPs 与手写算子库的区别在于组织方式:每个 op 先以一份 spec 声明,再由 agent 依据这份 spec 生成实现。spec 是代码生成的唯一依据,也是验收的标准: - -- 正确性以 spec 指定的参考实现为准; -- 性能以 roofline 模型给出的上界为准。 - -两项验收都不依赖人的判断。因此一个实现可以随时从 spec 重新生成,spec 却不能从实现反推。 - -对使用者而言,TileOPs 提供一批可以直接调用的 op。形状与 dtype 在调用时确定;特化后的 kernel 在首次使用时构造并缓存,之后可以与 CUDA graph 配合使用。每个 op 各自声明是否支持 `torch.compile(fullgraph=True)`。 +- **可维护**:每个 op 由一份 spec 声明,agent 依据 spec 生成实现。同一 family 的 op 共用一套接口与规则,新增的 op 和 kernel 遵循这套接口与规则。 +- **可验证**:spec 指定正确性所依据的参考实现。测试将 kernel 的输出与参考实现比较,性能评测将实测性能与 roofline 模型给出的上界比较。验收标准事先确定,检查过程自动执行。 +- **可调优**:roofline 模型给出每个 kernel 与性能上界之间的差距。[每晚的 benchmark](benchmarks/index.md) 在同一硬件上将每个 kernel 与最快的其他实现比较。 ## 安装 @@ -35,6 +30,7 @@ flops, nbytes = op.eval_roofline() # 本次调用所需的计算量与访存 ## 后续阅读 +- [博客](blog/index.md):TileOPs 开发中的技术探索。 - [使用指南](user-guide/index.md):读写 manifest、接入 `torch.compile`、benchmark 的计时方法、接入新硬件 backend。 - [API 参考](api/index.md):各 op family 的构造参数与调用方式。 - [性能数据](benchmarks/index.md):每晚在 H200 上实测,逐个 workload 与其他实现对比。 diff --git a/hooks.py b/hooks.py index 0c144ff2..285606dd 100644 --- a/hooks.py +++ b/hooks.py @@ -10,6 +10,8 @@ * mkdocs-static-i18n serves the default-language page where a locale has no translation; `on_page_markdown` prepends a notice, so a fallback page reads as a translation still to come rather than a broken one. +* A blog post's first paragraph is its subtitle; `on_page_content` marks it for + extra.css, so the post's Markdown carries no styling. """ from __future__ import annotations @@ -89,3 +91,15 @@ def on_config(config): if isinstance(section, dict) and "Benchmarks" in section: section["Benchmarks"] = entries return config + + +_FIRST_PARAGRAPH = re.compile(r"(\s*)

") + + +def on_page_content(html, page, config, files): + """Mark a blog post's subtitle.""" + src = page.file.src_path.replace("\\", "/") + name = src.split("/")[-1] + if not src.startswith("blog/") or name.startswith("index."): + return html + return _FIRST_PARAGRAPH.sub(r'\1

', html, count=1) diff --git a/mkdocs.yml b/mkdocs.yml index 97ac87ba..17cae28a 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -29,6 +29,8 @@ theme: nav: - Home: index.md + - Blog: + - blog/index.md - User Guide: - user-guide/index.md - Reading and Writing Specs: @@ -160,6 +162,7 @@ plugins: build: true nav_translations: Home: 首页 + Blog: 博客 Design: 设计文档 User Guide: 使用指南 Reading and Writing Specs: 读写 manifest From b47aa870b0e5f67c4af3316c1969dfbe75f29c4e Mon Sep 17 00:00:00 2001 From: lcy-seso Date: Fri, 2 Oct 2026 10:40:43 +0800 Subject: [PATCH 2/2] [Docs] Add a development guide, lay out the section indexes as cards, and reorder the Benchmarks and API Reference pages MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The development guide is the first User Guide page. It covers getting the code, the dev image and a local install, when kernels compile, the test tiers, when a benchmark is needed, and the title, description and CI of a pull request. It names no image tag or version that a release would change; it links where they are kept instead. The Chinese page is the source; the English page follows it. The User Guide and Design indexes group their pages by topic, and each page shows as a card with a one-line description. The Markdown stays a plain list per group; hooks.py marks the lists and extra.css draws the cards. The nav follows the same order, and the Chinese label for User Guide is now 用户指南. The timeline trace guide was the one page without a Chinese version. It mirrors TileOPs docs/perf/trace-timeline.md, so the translation follows that file. The Benchmarks pages, and the API Reference nav after them, run in the order a reader meets the ops: Elementwise and RoPE, the reductions and normalizations, Conv & Pool, GEMM and Quantization, Attention, MoE and Sampling, then the sequence-mixing kernels. The workload key above each table lists one label per line, with what it ran on under it. --- CLAUDE.md | 17 +- docs/api/index.md | 25 ++- docs/api/quantization.md | 2 +- docs/api/sampling.md | 17 +- docs/api/topk.md | 14 -- docs/assets/extra.css | 178 +++++++++++++--- docs/design/index.md | 35 +++- docs/index.zh.md | 2 +- .../memory-bound/global-memory-access.md | 4 +- .../memory-bound/global-memory-access.zh.md | 2 +- docs/performance-guides/trace-timeline.zh.md | 121 +++++++++++ docs/user-guide/development.md | 190 ++++++++++++++++++ docs/user-guide/development.zh.md | 152 ++++++++++++++ docs/user-guide/index.md | 36 +++- docs/user-guide/index.zh.md | 37 +++- hooks.py | 60 +++++- mkdocs.yml | 53 +++-- scripts/gen_bench_pages.py | 28 +-- 18 files changed, 834 insertions(+), 139 deletions(-) delete mode 100644 docs/api/topk.md create mode 100644 docs/performance-guides/trace-timeline.zh.md create mode 100644 docs/user-guide/development.md create mode 100644 docs/user-guide/development.zh.md diff --git a/CLAUDE.md b/CLAUDE.md index 7d541d3a..30d0e675 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -68,7 +68,7 @@ implementation of the same op on that workload. | Device time | Compare `device_busy_ms`, never wall-clock span | | Two questions | `Ratio`: is another kernel faster. `SOL`: how much faster the hardware allows anyone to go, its binding resource (`mem`/`comp`/`lat`) in a `Bound` column. Import the SOL arithmetic and thresholds from the checkout's roofline tool (M5); never re-derive them here | | Which page an op lands on | The manifest entry's `family:`, through `_MANIFEST_FAMILY` — an op TileOPs adds needs no change here. One the manifest does not declare falls back to its package, then to keywords | -| Page order | `DATA_PAGES` follows the API nav's order over the same families: one page per family except `Conv & Pool` (two) and `Other` (Top-k, FFT, mHC, Engram, the rest). `_BENCH_ORDER` in `hooks.py` repeats it — change one, change the other | +| Page order | `DATA_PAGES`: Elementwise, RoPE, Reduction, Normalization, Conv & Pool, GEMM, Quantization & Dequantization, Attention, MoE, Sampling, Linear Attention, SSM, Other. One page per family except `Conv & Pool` (two) and `Other` (FFT, mHC, Engram, the rest). `TopkSelectorFwdOp` declares `family: attention`, so its row is on Attention, while the API Reference documents it on the Sampling page. The API Reference nav follows it, with FFT, mHC and Engram after SSM and Top-k on the Sampling page. `_BENCH_ORDER` in `hooks.py` repeats it — change all three together | | Op order within a page | The order `docs/api/` names them, read by `api_op_order()`. An op no API page names comes last, ranked by verdict | | Rows follow the manifest | One row group per manifest label, one row per dtype under it in a `dtype` column. Labels keep the snapshot's order, which is the manifest's; the key above the table repeats it. A row no manifest describes takes its id, trailing dtype names split off, as its label | | Workload shapes | The snapshot names a workload but carries no shapes. `scripts/workload_shape.py` reads them from the spec manifest at the commit the benchmark ran on, joined by the `

/index.md` as a bare path with no title. Given a title it is promoted anyway and its sidebar row disappears. - Leave `toc.integrate` off: the page TOC renders in the right column, and it is diff --git a/docs/api/index.md b/docs/api/index.md index ad23c654..acff5727 100644 --- a/docs/api/index.md +++ b/docs/api/index.md @@ -13,30 +13,29 @@ op = GemmFwdOp() # construct once, reuse d = op(a, b) # the specialized kernel is built on first call ``` -The pages are ordered by how much an op composes: the pointwise transforms first, then -the axis reductions and the normalizations built on them, then the matmul and the -expert routing over it, then the windowed and spectral transforms, then the -sequence-model kernels built on all of the above. It is the order `tileops` declares -its op families in. The exception is Top-k, whose one op is exported from -`tileops.attention`. +The pages run in the order a reader meets the ops in a model: the pointwise +transforms and the positional rotation, then the axis reductions and the +normalizations built on them, the windowed kernels, the matmul and the +quantization around it, then attention, the expert routing after it and sampling, +then the sequence-model kernels. FFT, mHC and Engram follow, and Trace, +a tool rather than an op, comes last. The Benchmarks pages use the same order. | Page | What it covers | | --- | --- | | [Elementwise](elementwise.md) | unary and binary maps, activations, dropout, and the in-place forms | +| [RoPE](rope.md) | rotary position embedding — NeoX and interleaved layouts, Llama 3.1, YaRN, LongRoPE | | [Reduction](reduction.md) | sums, extrema, arg-reductions, cumulative scans, softmax | | [Normalization](normalization.md) | RMSNorm, LayerNorm, GroupNorm, BatchNorm and the fused variants | -| [Quantization](quantization.md) | INT8, FP8 and INT4 quantization, and INT8 dequantization | -| [Top-k](topk.md) | top-k selection | -| [GEMM](linear-algebra.md) | dense matmul — plain, batched, and the fp8 variants | | [Pooling](pool.md) | average, max and adaptive pooling, with and without indices, plus the chunked sequence mean | | [Convolution](convolution.md) | forward convolution over 1D, 2D and 3D inputs | -| [FFT](fft.md) | the discrete transform | -| [MoE](moe.md) | the routed mixture-of-experts FFN and its separately callable stages | -| [Sampling](sampling.md) | logits masks (top-k, top-p, min-p) and sampling, including chain speculative sampling | -| [RoPE](rope.md) | rotary position embedding — NeoX and interleaved layouts, Llama 3.1, YaRN, LongRoPE | +| [GEMM](linear-algebra.md) | dense matmul — plain, batched, and the fp8 variants | +| [Quantization & Dequantization](quantization.md) | INT8, FP8 and INT4 quantization, and INT8 dequantization | | [Attention](attention.md) | forward and backward attention, including the paged and decode kernels | +| [MoE](moe.md) | the routed mixture-of-experts FFN and its separately callable stages | +| [Top-k & Sampling](sampling.md) | top-k selection, logits masks (top-k, top-p, min-p) and sampling, including chain speculative sampling | | [Linear Attention](linear-attention.md) | DeltaNet, Gated DeltaNet and gated linear attention | | [Mamba](mamba.md) | the SSD scan, its decode step, and the chunked forms | +| [FFT](fft.md) | the discrete transform | | [mHC](mhc.md) | Manifold-Constrained Hyper-Connections — the pre/post pair around a layer | | [Engram](engram.md) | the Engram GateConv pair and its decode step | | [Trace](trace.md) | the in-kernel timeline tracer, a tool rather than an op | diff --git a/docs/api/quantization.md b/docs/api/quantization.md index abc25550..5a869add 100644 --- a/docs/api/quantization.md +++ b/docs/api/quantization.md @@ -1,4 +1,4 @@ -# Quantization Operators +# Quantization and Dequantization Operators Every op on this page is used the same way: construct it once, then call it. The constructor takes what the kernel is compiled with; the call takes the tensors. diff --git a/docs/api/sampling.md b/docs/api/sampling.md index ed530d2c..138a2998 100644 --- a/docs/api/sampling.md +++ b/docs/api/sampling.md @@ -1,13 +1,22 @@ -# Sampling Operators +# Top-k and Sampling Operators Every op on this page is used the same way: construct it once, then call it. The constructor takes what the kernel is compiled with; the call takes the tensors. Both are documented under each op — `__init__` and `forward`, where `forward` is what runs when you call `op(...)`. -The masks set every logit a filter drops to `-inf`, so a softmax over the result -renormalizes over what is kept. The samplers draw from a distribution, and take the -random seed and offset as tensors so a draw is reproducible. +The top-k selector returns the indices of the largest scores in each row. The masks +set every logit a filter drops to `-inf`, so a softmax over the result renormalizes +over what is kept. The samplers draw from a distribution, and take the random seed +and offset as tensors so a draw is reproducible. + +## Top-k selection + +::: tileops.attention.TopkSelectorFwdOp + options: + show_root_heading: true + heading_level: 3 + members: ["__init__", "forward"] ## Logits masks diff --git a/docs/api/topk.md b/docs/api/topk.md deleted file mode 100644 index 82207e5c..00000000 --- a/docs/api/topk.md +++ /dev/null @@ -1,14 +0,0 @@ -# Top-k Operators - -Every op on this page is used the same way: construct it once, then call it. The -constructor takes what the kernel is compiled with; the call takes the tensors. -Both are documented under each op — `__init__` and `forward`, where `forward` is -what runs when you call `op(...)`. - -## Top-k selection - -::: tileops.attention.TopkSelectorFwdOp - options: - show_root_heading: true - heading_level: 3 - members: ["__init__", "forward"] diff --git a/docs/assets/extra.css b/docs/assets/extra.css index a27779fa..e358e620 100644 --- a/docs/assets/extra.css +++ b/docs/assets/extra.css @@ -440,12 +440,12 @@ body, /* Every link in running text carries an arrow, so it reads as a link before it * is hovered: east for a page on this site, north-east for one that leaves it. - * Hover lifts a teal wash from the underline and nudges the arrow. Scoped to + * An absolute link back to this site counts as one on it. Hover lifts a teal wash from the underline and nudges the arrow. Scoped to * prose — paragraphs, list items, table cells — so headings' permalinks, the - * nav, buttons, badges and the Home card keep their own look, and no page's + * nav, buttons, badges and the page cards keep their own look, and no page's * Markdown has to mark a link for it. */ -.md-typeset :is(p, li, td, dd) a:not(:has(img)) { +.md-typeset :is(p, li, td, dd):not(.page-cards > li) a:not(.md-button, :has(img)) { padding: 0 0.08em; margin: 0 -0.08em; border-radius: 0.15em; @@ -453,11 +453,11 @@ body, transition: background-size 160ms ease, text-decoration-color 160ms ease; } -.md-typeset :is(p, li, td, dd) a:not(:has(img)):is(:hover, :focus-visible) { +.md-typeset :is(p, li, td, dd):not(.page-cards > li) a:not(.md-button, :has(img)):is(:hover, :focus-visible) { background-size: 100% 100%; } -.md-typeset :is(p, li, td, dd) a:not(:has(img))::after { +.md-typeset :is(p, li, td, dd):not(.page-cards > li) a:not(.md-button, :has(img))::after { content: ""; display: inline-block; width: 0.62em; @@ -471,18 +471,18 @@ body, transition: transform 160ms ease, opacity 160ms ease; } -.md-typeset :is(p, li, td, dd) a[href^="http"]:not(:has(img))::after { +.md-typeset :is(p, li, td, dd):not(.page-cards > li) a[href^="http"]:not([href^="https://tile-ai.github.io/TileOPs.github.io/"], .md-button, :has(img))::after { vertical-align: 0.12em; -webkit-mask-image: url("data:image/svg+xml,%3Csvg%20xmlns='http://www.w3.org/2000/svg'%20viewBox='0%200%2012%2012'%3E%3Cpath%20d='M4%202h6v6M10%202%202.5%209.5'%20fill='none'%20stroke='%23000'%20stroke-width='1.6'%20stroke-linecap='round'%20stroke-linejoin='round'/%3E%3C/svg%3E"); mask-image: url("data:image/svg+xml,%3Csvg%20xmlns='http://www.w3.org/2000/svg'%20viewBox='0%200%2012%2012'%3E%3Cpath%20d='M4%202h6v6M10%202%202.5%209.5'%20fill='none'%20stroke='%23000'%20stroke-width='1.6'%20stroke-linecap='round'%20stroke-linejoin='round'/%3E%3C/svg%3E"); } -.md-typeset :is(p, li, td, dd) a:not(:has(img)):is(:hover, :focus-visible)::after { +.md-typeset :is(p, li, td, dd):not(.page-cards > li) a:not(.md-button, :has(img)):is(:hover, :focus-visible)::after { opacity: 1; transform: translateX(0.1em); } -.md-typeset :is(p, li, td, dd) a[href^="http"]:not(:has(img)):is(:hover, :focus-visible)::after { +.md-typeset :is(p, li, td, dd):not(.page-cards > li) a[href^="http"]:not([href^="https://tile-ai.github.io/TileOPs.github.io/"], .md-button, :has(img)):is(:hover, :focus-visible)::after { transform: translate(0.08em, -0.08em); } @@ -1345,13 +1345,14 @@ html[lang="zh"] .md-nav__title { } } -/* Labels in one column, what varies in the next. A label past the column's - * width wraps at its hyphens, so a long one never pushes the values off. +/* One column: each label on its own line, what it ran on under it. Two columns + * put a label and its shapes side by side, and the eye then reads across a gap + * that changes width from one key to the next. */ .md-typeset .wl-key ul.wl-rows { display: grid; - grid-template-columns: fit-content(24em) minmax(0, 1fr); - gap: 0.1em 0.9em; + grid-template-columns: minmax(0, 1fr); + gap: 0; align-items: baseline; margin: 0; padding: 0; @@ -1362,11 +1363,12 @@ html[lang="zh"] .md-nav__title { display: contents; } -/* No width for two columns: each label on its own line, what varies under it. */ -@media (max-width: 50rem) { - .md-typeset .wl-key ul.wl-rows { - grid-template-columns: minmax(0, 1fr); - } +.md-typeset .wl-key ul.wl-rows .wl-id { + margin-top: 0.4em; +} + +.md-typeset .wl-key ul.wl-rows li:first-child .wl-id { + margin-top: 0; } /* A hairline between the label and what it ran on — the same rule the table @@ -1389,20 +1391,6 @@ html[lang="zh"] .md-nav__title { min-width: 0; } -/* Labels too long for the label column: each on its own line, values under it. */ -.md-typeset .wl-key ul.wl-rows.wl-long { - grid-template-columns: minmax(0, 1fr); - row-gap: 0; -} - -.md-typeset .wl-key ul.wl-rows.wl-long .wl-id { - margin-top: 0.4em; -} - -.md-typeset .wl-key ul.wl-rows.wl-long li:first-child .wl-id { - margin-top: 0; -} - .md-typeset .wl-key .wl-delta:empty { border-left: 0; } @@ -1730,3 +1718,131 @@ html[lang="zh"] .md-nav__title { font-weight: 600; line-height: 1.35; } + +/* A section index's page list, drawn as cards by topic. `hooks.py` marks the + * lists; each item is a link, which becomes the card's title, and a line of + * description under it. Groups alternate violet and teal: a faint fill of that + * colour, a rule along the card's top, a numbered kicker, and a small corner of + * tiles that fills in on hover. Two columns where the page + * has the width, one below. + */ +.md-typeset:has(> ul.page-cards) { + counter-reset: page-card; +} + +.md-typeset ul.page-cards { + --card-accent: var(--tf-violet); + --card-tint: rgba(90, 62, 133, 0.08); + display: grid; + grid-template-columns: repeat(auto-fill, minmax(17rem, 1fr)); + gap: 1rem; + margin: 0.9rem 0 1.8rem; + padding: 0; + list-style: none; +} + +.md-typeset ul.page-cards:nth-of-type(even) { + --card-accent: var(--tf-teal); + --card-tint: rgba(22, 112, 107, 0.08); +} + +.md-typeset ul.page-cards > li { + position: relative; + overflow: hidden; + min-height: 7.2rem; + margin: 0; + padding: 1.05rem 1.2rem 1.9rem; + border: 1px solid var(--tf-line); + border-top: 3px solid var(--card-accent); + border-radius: 0.6rem; + background: var(--card-tint); + box-shadow: 0 1px 2px rgba(27, 19, 39, 0.05); + color: var(--tf-muted); + font-size: 0.74rem; + line-height: 1.6; + counter-increment: page-card; + transition: border-color 180ms ease, box-shadow 180ms ease, transform 180ms ease; +} + +/* The kicker: the card's number in the index, in the group's accent. */ +.md-typeset ul.page-cards > li::before { + content: counter(page-card, decimal-leading-zero); + display: block; + margin-bottom: 0.45rem; + color: var(--card-accent); + font-family: var(--tf-mono); + font-size: 0.62rem; + font-weight: 700; + letter-spacing: 0.12em; +} + +/* A corner of tiles, faint until the card is hovered. */ +.md-typeset ul.page-cards > li::after { + content: ""; + position: absolute; + top: 0.9rem; + right: 0.9rem; + width: 46px; + height: 46px; + background-color: var(--card-accent); + -webkit-mask: url("data:image/svg+xml,%3Csvg%20xmlns='http://www.w3.org/2000/svg'%20width='46'%20height='46'%3E%3Crect%20x='0'%20y='0'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.9'/%3E%3Crect%20x='12'%20y='0'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.5'/%3E%3Crect%20x='0'%20y='12'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.5'/%3E%3Crect%20x='24'%20y='0'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.25'/%3E%3Crect%20x='12'%20y='12'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.3'/%3E%3Crect%20x='0'%20y='24'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.25'/%3E%3Crect%20x='36'%20y='0'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.12'/%3E%3Crect%20x='24'%20y='12'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.15'/%3E%3Crect%20x='12'%20y='24'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.15'/%3E%3Crect%20x='0'%20y='36'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.12'/%3E%3C/svg%3E") right top / contain no-repeat; + mask: url("data:image/svg+xml,%3Csvg%20xmlns='http://www.w3.org/2000/svg'%20width='46'%20height='46'%3E%3Crect%20x='0'%20y='0'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.9'/%3E%3Crect%20x='12'%20y='0'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.5'/%3E%3Crect%20x='0'%20y='12'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.5'/%3E%3Crect%20x='24'%20y='0'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.25'/%3E%3Crect%20x='12'%20y='12'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.3'/%3E%3Crect%20x='0'%20y='24'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.25'/%3E%3Crect%20x='36'%20y='0'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.12'/%3E%3Crect%20x='24'%20y='12'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.15'/%3E%3Crect%20x='12'%20y='24'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.15'/%3E%3Crect%20x='0'%20y='36'%20width='10'%20height='10'%20rx='1.5'%20fill='%23000'%20fill-opacity='0.12'/%3E%3C/svg%3E") right top / contain no-repeat; + opacity: 0.18; + transition: opacity 180ms ease, transform 180ms ease; + pointer-events: none; +} + +.md-typeset ul.page-cards > li:hover, +.md-typeset ul.page-cards > li:focus-within { + border-color: var(--card-accent); + box-shadow: 0 14px 30px -18px rgba(27, 19, 39, 0.45); + transform: translateY(-3px); +} + +.md-typeset ul.page-cards > li:hover::after, +.md-typeset ul.page-cards > li:focus-within::after { + opacity: 0.6; + transform: scale(1.06); +} + +/* The title's link covers the whole card, so the card is one click target. */ +.md-typeset ul.page-cards > li > a:first-child { + display: block; + max-width: calc(100% - 3.2rem); + margin-bottom: 0.35rem; + color: var(--md-default-fg-color); + font-family: var(--tf-display); + font-size: 0.95rem; + font-weight: 700; + line-height: 1.3; + text-decoration: none; +} + +.md-typeset ul.page-cards > li > a:first-child::before { + content: ""; + position: absolute; + inset: 0; + z-index: 1; + border-radius: inherit; +} + +/* The arrow sits in the card's bottom-right corner, so a title that wraps never + * leaves it alone on a line. + */ +.md-typeset ul.page-cards > li > a:first-child::after { + content: "→"; + position: absolute; + right: 1.2rem; + bottom: 0.9rem; + color: var(--card-accent); + font-size: 1rem; + transition: transform 180ms ease; +} + +.md-typeset ul.page-cards > li:hover > a:first-child::after { + transform: translateX(0.25em); +} + +html[lang="zh"] .md-typeset ul.page-cards > li > a:first-child { + font-weight: 700; +} diff --git a/docs/design/index.md b/docs/design/index.md index 015864f0..42e8e830 100644 --- a/docs/design/index.md +++ b/docs/design/index.md @@ -4,11 +4,30 @@ Architecture and design documentation for TileOPs internals. The pages below mirror `docs/design/` in the [`tile-ai/TileOPs`](https://github.com/tile-ai/TileOPs) repository — the source of truth — pulled in at site build time. -- [Architecture](architecture.md) — top-level module layout and the spec-driven pipeline. -- [Op Manifest](manifest.md) — the `src/tileops/manifest/` package as the source of truth for op interfaces. -- [Op Interface Design](ops-design.md) — playbook for scaffolding a new op from a manifest entry. -- [Op Interface Reference](ops-design-reference.md) — interface tables, codegen, naming, and the family-base protocol. -- [Slot Rules](op-slot-rules.md) — the authoritative rule, example, and common mistakes per op-file slot. -- [Roofline](roofline.md) — performance model and the `roofline` manifest field. -- [Testing & Benchmarking](testing.md) — separation of correctness tests and profiling benchmarks. -- [Layer Boundaries](layer-boundaries.md) — what each layer owns and depends on, and the interfaces layers compose through. +## System structure + +- [Architecture](architecture.md) + Top-level module layout and the spec-driven pipeline. +- [Layer Boundaries](layer-boundaries.md) + What each layer owns and depends on, and the interfaces layers compose through. + +## The spec + +- [Op Manifest](manifest.md) + The `src/tileops/manifest/` package as the source of truth for op interfaces. + +## Building an op + +- [Op Interface Design](ops-design.md) + Playbook for scaffolding a new op from a manifest entry. +- [Op Interface Reference](ops-design-reference.md) + Interface tables, codegen, naming, and the family-base protocol. +- [Slot Rules](op-slot-rules.md) + The authoritative rule, example, and common mistakes per op-file slot. + +## Acceptance + +- [Testing & Benchmarking](testing.md) + Separation of correctness tests and profiling benchmarks. +- [Roofline](roofline.md) + Performance model and the `roofline` manifest field. diff --git a/docs/index.zh.md b/docs/index.zh.md index d3d770d6..f0fe0df5 100644 --- a/docs/index.zh.md +++ b/docs/index.zh.md @@ -31,7 +31,7 @@ flops, nbytes = op.eval_roofline() # 本次调用所需的计算量与访存 ## 后续阅读 - [博客](blog/index.md):TileOPs 开发中的技术探索。 -- [使用指南](user-guide/index.md):读写 manifest、接入 `torch.compile`、benchmark 的计时方法、接入新硬件 backend。 +- [用户指南](user-guide/index.md):读写 manifest、接入 `torch.compile`、benchmark 的计时方法、接入新硬件 backend。 - [API 参考](api/index.md):各 op family 的构造参数与调用方式。 - [性能数据](benchmarks/index.md):每晚在 H200 上实测,逐个 workload 与其他实现对比。 diff --git a/docs/performance-guides/memory-bound/global-memory-access.md b/docs/performance-guides/memory-bound/global-memory-access.md index 45631d6b..8966d00d 100644 --- a/docs/performance-guides/memory-bound/global-memory-access.md +++ b/docs/performance-guides/memory-bound/global-memory-access.md @@ -6,8 +6,8 @@ choosing one. ## Checking whether DRAM bandwidth is the current limit {#regime} -[Elementwise](https://tile-ai.github.io/TileOPs.github.io/api/elementwise/) and -[Reduction](https://tile-ai.github.io/TileOPs.github.io/api/reduction/) are the +[Elementwise](../../api/elementwise.md) and +[Reduction](../../api/reduction.md) are the typical memory-bound kernels. Each recommendation on this page states when it applies, why it applies, and what the wrong and right code look like. diff --git a/docs/performance-guides/memory-bound/global-memory-access.zh.md b/docs/performance-guides/memory-bound/global-memory-access.zh.md index f10cc19d..4893d8e3 100644 --- a/docs/performance-guides/memory-bound/global-memory-access.zh.md +++ b/docs/performance-guides/memory-bound/global-memory-access.zh.md @@ -4,7 +4,7 @@ ## 确认 DRAM 带宽是否为当前的瓶颈 {#regime} -[Elementwise](https://tile-ai.github.io/TileOPs.github.io/api/elementwise/) 与 [Reduction](https://tile-ai.github.io/TileOPs.github.io/api/reduction/) 是典型的访存受限 kernel。本页每条建议都写明触发条件、成因,以及反例与正例代码。 +[Elementwise](../../api/elementwise.md) 与 [Reduction](../../api/reduction.md) 是典型的访存受限 kernel。本页每条建议都写明触发条件、成因,以及反例与正例代码。 本页的实测都在同一组条件下取得:**输入大于 L2 的 60 MiB,且 block 数足以填满整卡**(H200 有 132 个 SM)。此时 DRAM 带宽是主要瓶颈,访存模式的差别直接反映在性能上。 diff --git a/docs/performance-guides/trace-timeline.zh.md b/docs/performance-guides/trace-timeline.zh.md new file mode 100644 index 00000000..1958dd11 --- /dev/null +++ b/docs/performance-guides/trace-timeline.zh.md @@ -0,0 +1,121 @@ +# kernel 内的时间线追踪 { #in-kernel-timeline-trace } + +[`tileops.trace`](../api/trace.md) 是一个在 kernel 内部记录时间线的追踪工具,用于诊断 kernel 的性能。它在 kernel 内为每个 CTA 记录时间戳,并渲染成一条可以滚动查看的时间线,显示执行区间、空隙,以及 producer 与 consumer 之间的重叠。这些信息是 `ncu` 这类以 kernel 为单位的 profiler 看不到的。warp specialization 的 kernel 最能用上它:这类 kernel 的关键就是让 producer(TMA)与 consumer(WGMMA)两个 warpgroup 的执行相互重叠。 + +## 工作方式 { #how-it-works } + +1. 在 kernel 代码中用标记(`trace.range`、`trace.group` 等)标注要记录的区间。 +1. 标记**总是以占位的形式生成**。构建时,kernel 要么被 **lower**:标记变成真正调用 `clock64()` 记录时间的代码,并在输出末尾多出一个 `slots`;要么被 **strip**:标记变成空操作,生成的 CUDA 与未插桩的版本完全相同。 +1. 运行时,`trace.run` 执行 kernel,解码 `slots` 缓冲区,并写出一个自包含的 Plotly HTML 时间线。 +1. 由进程内的开关 `trace.enable()` 决定 lower 还是 strip,因此关闭时追踪**没有任何开销**,标记可以留在生产代码中。 + +时间戳来自 `clock64()`,即每个 SM 的周期计数器。 + +## 编写带追踪的 kernel { #write-a-traced-kernel } + +下面是一个完整的、接入了追踪的 warp specialization GEMM(示意用,单缓冲)。编号 `(1)`–`(7)` 的标记是仅有的与追踪相关的代码,各自的说明和 API 文档链接见代码下方。生产环境中的多 stage 版本见 [`src/tileops/kernels/gemm/dense.py`](https://github.com/tile-ai/TileOPs/blob/main/src/tileops/kernels/gemm/dense.py)。 + +```{ .python .annotate } +import functools +import tilelang +import tilelang.language as T +from tileops.trace import trace # (1)! + + +@functools.lru_cache(maxsize=32) +def build_gemm(m, n, k, dtype="float16", traced=False): + @tilelang.jit(out_idx=trace.out_idx(1, traced)) # (2)! + def factory(block_m=128, block_n=128, block_k=64): + @T.prim_func + def main(a: T.Tensor((m, k), dtype), b: T.Tensor((n, k), dtype), + c: T.Tensor((m, n), dtype)): + with T.Kernel(T.ceildiv(n, block_n), T.ceildiv(m, block_m), + threads=256) as (bx, by): + a_smem = T.alloc_shared((block_m, block_k), dtype) + b_smem = T.alloc_shared((block_n, block_k), dtype) + c_local = T.alloc_fragment((block_m, block_n), "float") + full = T.alloc_barrier(128) + tx = T.get_thread_binding() + + if tx < 128: + with trace.group("producer", lead=0): # (3)! + for ki in T.serial(T.ceildiv(k, block_k)): + with trace.range("tma", lane="tma"): # (4)! + T.tma_copy(a[by * block_m, ki * block_k], a_smem, barrier=full) + T.tma_copy(b[bx * block_n, ki * block_k], b_smem, barrier=full) + with trace.range("arrive", lane="barrier"): + T.barrier_arrive(full) + else: + with trace.group("consumer", lead=128): + T.clear(c_local) + for ki in T.serial(T.ceildiv(k, block_k)): + with trace.range("wait", lane="barrier"): + T.barrier_wait(full, ki % 2) + with trace.range("mma", lane="wgmma"): # (5)! + T.wgmma_gemm(a_smem, b_smem, c_local, transpose_B=True) + with trace.range("epilogue"): + T.copy(c_local, c[by * block_m, bx * block_n]) + + trace.dag("arrive", "wait") # (6)! + return trace.finalize(main, traced=traced, max_events=1024) # (7)! + return factory +``` + +1. 导入 trace 命名空间。下面的每个调用都是这个 `trace` 对象上的方法,完整说明见 [API 参考](../api/trace.md)。 +1. [`trace.out_idx(n_outputs, traced)`](../api/trace.md#tileops.trace.api._Trace.out_idx) 给出 `@tilelang.jit` 的 `out_idx`。只有 `traced` 时它才多出一个位置给末尾的 `slots` 输出,因此同一个构建函数在开启和关闭追踪时都能使用。 +1. [`trace.group(name, lead)`](../api/trace.md#tileops.trace.api._Trace.group) 声明由哪个 warpgroup 记录。`lead` 是被选出的写入线程(`tx == lead`):计算仍在所有线程上执行,只有时间戳由 `lead` 写入。 +1. [`trace.range(name, lane)`](../api/trace.md#tileops.trace.api._Trace.range) 是一个 `with` 块,从进入计时到退出,在子行 `lane` 上画成一段条形。控制流放不进 `with` 时,用 [`trace.range_start`](../api/trace.md#tileops.trace.api._Trace.range_start) 与 [`trace.range_end`](../api/trace.md#tileops.trace.api._Trace.range_end);只需要一个零宽度的时间点时,用 [`trace.record`](../api/trace.md#tileops.trace.api._Trace.record)。 +1. lane 的名字(`"tma"`、`"barrier"`、`"wgmma"`,以及默认的 `"main"`)就是时间线上的各行。 +1. [`trace.dag(src, dst)`](../api/trace.md#tileops.trace.api._Trace.dag) 声明从一个命名区间指向另一个区间的依赖箭头(`arrive` → `wait`),每出现一次画一条。 +1. [`trace.finalize(func, traced, max_events)`](../api/trace.md#tileops.trace.api._Trace.finalize) 在 `traced` 时 lower 标记并加上 `slots` 输出,否则把标记 strip 成零开销。`traced` **必须是构建函数缓存键的一部分**,这样同一形状下带追踪和不带追踪的两次构建才不会冲突。 + +## 运行带追踪的 kernel { #running-a-traced-kernel } + +调用构建函数时传入 `traced=trace.enabled`,再把编译好的 kernel 交给 `trace.run`。同一个 `forward` 在两种模式下都能使用:追踪关闭时原样返回输出;追踪开启时写出时间线,并且只返回真正的输出。因此调用方不需要自己按开关分支。 + +```{ .python .annotate } +def forward(self, a, b): + compiled = build_gemm(self.m, self.n, self.k, self.dtype_str, + traced=trace.enabled)(**self.config) # (1)! + return trace.run(compiled, (a, b), stem="gemm_128x256x512") # (2)! +``` + +1. 按开关构建对应的 kernel:[`trace.enabled`](../api/trace.md#tileops.trace.api._Trace.enabled) 决定选用带追踪的版本还是 strip 后的版本,它也是标记 `(7)` 所说的缓存键的一部分。 +1. [`trace.run(compiled, inputs, stem=...)`](../api/trace.md#tileops.trace.api._Trace.run) 运行 kernel。带追踪时,它把末尾的 `slots` 拆出来解码,写出 `debug/.html`,每次调用都生成一个不会重名的新文件。它内部由 [`decode`](../api/trace.md#tileops.trace.api._Trace.decode) 和 [`dump`](../api/trace.md#tileops.trace.api._Trace.dump) 组成,两者也可以直接调用。 + +## 开启追踪 { #enabling-tracing } + +追踪默认关闭。在程序启动时打开一次进程内的开关,之后照常运行: + +```{ .python .annotate } +from tileops.trace import trace + +trace.enable() # (1)! +c = op.forward(a, b) # (2)! +``` + +1. [`trace.enable(output="debug")`](../api/trace.md#tileops.trace.api._Trace.enable) 打开追踪,并指定输出目录,默认是 `debug/`(已被 gitignore)。这个开关只在当前进程内有效:不读取环境变量,也不对 `tilelang` 做 monkeypatch。相关的还有 [`trace.disable()`](../api/trace.md#tileops.trace.api._Trace.disable)、[`trace.enabled`](../api/trace.md#tileops.trace.api._Trace.enabled) 与 [`trace.output`](../api/trace.md#tileops.trace.api._Trace.output)。 +1. 此后运行的每个带追踪的 kernel 都会写出 `debug/.html`。 + +在 pytest 中,`--trace-kernel` 会在任何 kernel 构建之前,由 `pytest_configure` 调用 `trace.enable()`: + +```bash +pytest tests/ops/test_gemm.py --trace-kernel +``` + +## 阅读时间线 { #reading-the-timeline } + + + +- **顶部的 CTA 标签页**:每个 CTA(block)一条时间线。 +- **lane**(各行)来自代码中 `group` 与 `lane` 的名字,例如 `producer / tma`、`consumer / wgmma`。每个 `range` 是一段条形,鼠标悬停可以看到它的名字与周期区间。 +- **横轴**是 SM 的原始周期数(`clock64()`),每个 CTA 从零开始。 +- **箭头**是代码中声明的 `dag` 边(例如 producer 的 `arrive` → consumer 的 `wait`),每出现一次画一条。从箭头可以读出交接的延迟,以及 consumer 是否在空等。 +- 缩放与平移只在水平方向进行。 + +需要留意的现象: + +- `wgmma` 这一行上有空隙:consumer 停下来等待 TMA 加载。 +- 相邻迭代之间的 `dag` 箭头没有重叠:没有形成流水。 +- 某一行明显比其他行长:负载不均衡。 diff --git a/docs/user-guide/development.md b/docs/user-guide/development.md new file mode 100644 index 00000000..40c585e5 --- /dev/null +++ b/docs/user-guide/development.md @@ -0,0 +1,190 @@ +# Development Guide { #development-guide } + +## Getting the code { #get-code } + +```bash +git clone https://github.com//TileOPs +cd TileOPs +git remote add upstream https://github.com/tile-ai/TileOPs +git fetch upstream +git switch -c upstream/main +``` + +`` is the GitHub account that holds the fork of +[tile-ai/TileOPs](https://github.com/tile-ai/TileOPs). A development branch starts +from the latest upstream `main`, and the upstream repository is recorded as +`upstream`. + +## Setting up an environment { #setup } + +### The dev image { #docker } + +```bash +docker run --rm -it --gpus all \ + -v "$(pwd)":/workspace -w /workspace \ + ghcr.io/tile-ai/tileops-runner: + +# inside the container +pip install -e . --no-deps --no-build-isolation +``` + +- The dev images are published at + [ghcr.io/tile-ai/tileops-runner](https://github.com/tile-ai/TileOPs/pkgs/container/tileops-runner); + development uses a tag ending in `-dev`. +- The image carries CUDA, PyTorch, TileLang and the test tools, and comes from the + same build as the CI's GPU runners. +- `--no-deps` skips resolving and installing dependencies, and uses the ones the + image already has. + +The dev image does not include pre-commit. Run the following on the host, or in +any other environment with Python: + +```bash +pip install pre-commit +pre-commit install +``` + +### A local environment { #local } + +```bash +pip install -e '.[dev]' -c constraints.txt +pre-commit install +``` + +- The supported combination of Python, PyTorch, CUDA, GPU architecture and + TileLang is the one listed under Prerequisites in the TileOPs + [README](https://github.com/tile-ai/TileOPs#installation). +- `-c constraints.txt` applies the repository's dependency constraints, so the + local dependencies match the combination CI validates. + +### Checking the environment { #verify } + +```bash +python -m pytest -q tests -m smoke +``` + +Each kernel under test is compiled and cached on its first call, and later +identical calls reuse the cache; see [Kernel compilation and caching](#compile). + +## Making a change { #change } + +### The relevant spec and design docs { #design-first } + +The design docs and the manifest define the constraints on ops, kernels and +tests. When a change to the implementation affects those constraints, the same PR +updates the spec. + +| Change | Documents | +| --- | --- | +| Adding an op | [Adding a new op](../new-op.md) | +| An existing op or kernel | [Op Interfaces](../design/ops-design.md), [Slot Rules](../design/op-slot-rules.md) | +| A spec in the manifest | [Spec Fields](manifest/writing.md), [Manifest](../design/manifest.md) | +| Tests | [Testing](../design/testing.md), [Layer Boundaries](../design/layer-boundaries.md) | +| An op's docstring | [Docstrings](https://github.com/tile-ai/TileOPs/blob/main/docs/development.md#docstrings) | + +### Kernel compilation and caching { #compile } + +```python +op = GemmFwdOp() +d = op(a, b) # first call: compiles the kernel and caches it +d = op(a, b) # the same call: reuses the cache +``` + +- Installing TileOPs compiles no kernels. TileLang compiles a kernel the first time + an op is called. +- An op looks up the entry serving a call by the whole call; what is compiled is + cached by build identity, and calls that select the same implementation class + with the same build identity share one entry. TileLang compiles when no entry + matches. Build identity is defined in + [How an op selects a kernel](dispatch/index.md). +- Under an editable install, changes to Python source take effect directly; a + change to dependencies or build configuration needs the install command run + again. + +## Running the tests { #tests } + +| Command | Tests included | +| --- | --- | +| `python -m pytest -q tests -m smoke` | `smoke`: the critical path | +| `python -m pytest -q tests -m "smoke or full"` | adds `full`: standard correctness coverage | +| `python -m pytest -q tests -m "smoke or full or nightly"` | adds `nightly`: exhaustive and long-running cases | +| `python -m pytest -q tests` | every test, with no marker filter | + +A single test file runs with `python -m pytest -q `. + +These checks need no GPU: + +```bash +python -m pytest -q tests/test_validate_manifest.py # manifest spec validation +python -m pytest -q benchmarks/tests # benchmark infrastructure tests +pre-commit run --all-files # lint, the same as CI's pre-commit check +``` + +## Running benchmarks { #bench } + +```bash +PIP_NO_BUILD_ISOLATION=1 pip install -e '.[dev,bench]' -c constraints.txt +python -m pytest -q +``` + +- A PR that changes a kernel or an op includes benchmark results, compared against + an implementation outside TileOPs. +- The baseline libraries install through the `bench` extra. The dev image installs + them at build time too; a baseline that fails to install there does not stop the + build, and `sgl-kernel` is not in the image. A baseline missing from the + container is installed at the version the repository declares. +- How the numbers are timed: [How a benchmark is timed](../timing.md). + +## Opening a PR { #pr } + +### Title { #pr-title } + +| Format | Example | +| --- | --- | +| `[Type] ` | `[Doc] Fix the install command in the README` | +| `[Type][Scope] ` | `[BugFix][Elementwise] Build the floored tiers at every tuned fold width` | +| `[Type][foundry][Scope] ` | `[Perf][foundry][Elementwise] Take the floored tier's reciprocal from one MUFU instruction` | + +- CI checks the PR title's format. The values of `Type` are defined in + [`.claude/conventions/types.sh`](https://github.com/tile-ai/TileOPs/blob/main/.claude/conventions/types.sh). +- `foundry` marks a PR whose kernels were generated by + [TileFoundry](https://github.com/tile-ai/TileFoundry). + +### Description { #pr-body } + +The PR description follows the +[PR template](https://github.com/tile-ai/TileOPs/blob/main/.github/PULL_REQUEST_TEMPLATE.md). +When a change touches `tests/`, the description includes the change in the test +count: + +```bash +python scripts/test_node_delta.py --base upstream/main +``` + +### CI { #ci } + +- A draft PR runs the PR title check and the manifest statistics only, and skips + the CPU checks and GPU smoke tests below. They run once the PR is marked + **Ready for review**. +- The CPU checks are pre-commit, gitleaks, manifest validation, actionlint, the + compile contract check and the packaging check, plus the benchmark contract tests + when a change touches the benchmarks. +- The GPU smoke tests run once pre-commit, gitleaks and actionlint pass, scoped by + the files the PR changes. A PR that changes no Python, manifest or native source + skips them. + +## FAQ { #faq } + +### A local install fails while re-resolving CUDA or TileLang { #faq-rebuild } + +```bash +PIP_NO_BUILD_ISOLATION=1 pip install -e '.[dev]' -c constraints.txt +``` + +With CUDA and TileLang already installed, turning off build isolation lets the +build use the installed versions. + +### Checks that run without an SM90 GPU { #faq-no-gpu } + +The three GPU-free checks under [Running the tests](#tests). The tests that need a +GPU run in the PR's CI. diff --git a/docs/user-guide/development.zh.md b/docs/user-guide/development.zh.md new file mode 100644 index 00000000..3cff28ad --- /dev/null +++ b/docs/user-guide/development.zh.md @@ -0,0 +1,152 @@ +# 开发指南 { #development-guide } + +## 获取代码 { #get-code } + +```bash +git clone https://github.com//TileOPs +cd TileOPs +git remote add upstream https://github.com/tile-ai/TileOPs +git fetch upstream +git switch -c upstream/main +``` + +`` 是 fork [tile-ai/TileOPs](https://github.com/tile-ai/TileOPs) 所在的 GitHub 账号。开发分支从上游最新的 `main` 创建,上游仓库记录为 `upstream`。 + +## 搭建开发环境 { #setup } + +### 使用 dev 镜像 { #docker } + +```bash +docker run --rm -it --gpus all \ + -v "$(pwd)":/workspace -w /workspace \ + ghcr.io/tile-ai/tileops-runner: + +# 以下命令在容器内执行 +pip install -e . --no-deps --no-build-isolation +``` + +- dev 镜像发布在 [ghcr.io/tile-ai/tileops-runner](https://github.com/tile-ai/TileOPs/pkgs/container/tileops-runner),开发使用以 `-dev` 结尾的 tag。 +- 镜像包含 CUDA、PyTorch、TileLang 与测试工具,与 CI 的 GPU runner 出自同一构建流程。 +- `--no-deps` 跳过依赖的解析与安装,直接使用镜像中已有的依赖。 + +dev 镜像不包含 pre-commit。以下命令在宿主机或另一个已安装 Python 的环境中执行: + +```bash +pip install pre-commit +pre-commit install +``` + +### 使用本地环境 { #local } + +```bash +pip install -e '.[dev]' -c constraints.txt +pre-commit install +``` + +- 支持的 Python、PyTorch、CUDA、GPU 架构与 TileLang 版本组合,以 TileOPs 仓库 [README](https://github.com/tile-ai/TileOPs#installation) 的 Prerequisites 为准。 +- `-c constraints.txt` 使用仓库提供的依赖约束,使本地依赖与 CI 验证的组合一致。 + +### 确认环境可用 { #verify } + +```bash +python -m pytest -q tests -m smoke +``` + +被测的 kernel 在首次调用时编译并写入缓存,之后相同的调用复用缓存,见 [kernel 的编译与缓存](#compile)。 + +## 修改代码 { #change } + +### 相关的 spec 与设计文档 { #design-first } + +设计文档与 manifest 定义 op、kernel 与测试的约束。实现的变更影响这些约束时,同一个 PR 同步更新对应的 spec。 + +| 改动内容 | 相关文档 | +| --- | --- | +| 新增 op | [添加新 op](../new-op.md) | +| 修改已有的 op 或 kernel | [Op 接口](../design/ops-design.md)、[Slot 规则](../design/op-slot-rules.md) | +| 修改 manifest 中的 spec | [写一个 spec](manifest/writing.md)、[manifest 规范](../design/manifest.md) | +| 修改测试 | [测试](../design/testing.md)、[层间边界](../design/layer-boundaries.md) | +| 修改 op 的 docstring | [docstring 的写法](https://github.com/tile-ai/TileOPs/blob/main/docs/development.md#docstrings) | + +### kernel 的编译与缓存 { #compile } + +```python +op = GemmFwdOp() +d = op(a, b) # 首次调用:编译 kernel 并写入缓存 +d = op(a, b) # 相同的调用:复用缓存 +``` + +- 安装 TileOPs 时不编译 kernel。kernel 在 op 首次被调用时由 TileLang 编译。 +- op 按完整的调用信息查找服务这次调用的 entry;编译结果按 build identity 缓存,选中同一实现类且 build identity 相同的调用共用同一个 entry。没有对应的 entry 时,TileLang 执行编译。build identity 的定义见 [op 如何选择 kernel](dispatch/index.md)。 +- editable 安装下,Python 源码的改动直接生效;依赖或构建配置变更后,需要重新执行安装命令。 + +## 运行测试 { #tests } + +| 命令 | 包含的测试 | +| --- | --- | +| `python -m pytest -q tests -m smoke` | `smoke`:关键路径 | +| `python -m pytest -q tests -m "smoke or full"` | 另含 `full`:标准的正确性覆盖 | +| `python -m pytest -q tests -m "smoke or full or nightly"` | 另含 `nightly`:穷举与耗时长的用例 | +| `python -m pytest -q tests` | 全部测试,不按 marker 过滤 | + +单个测试文件以 `python -m pytest -q ` 运行。 + +以下检查不需要 GPU: + +```bash +python -m pytest -q tests/test_validate_manifest.py # manifest spec 校验 +python -m pytest -q benchmarks/tests # benchmark 基础设施测试 +pre-commit run --all-files # 代码检查,与 CI 中的 pre-commit 一项相同 +``` + +## 运行 benchmark { #bench } + +```bash +PIP_NO_BUILD_ISOLATION=1 pip install -e '.[dev,bench]' -c constraints.txt +python -m pytest -q +``` + +- 改动 kernel 或 op 的 PR 需附上 benchmark 结果,对比对象是 TileOPs 以外的实现。 +- benchmark 的对比库通过 `bench` extra 安装。dev 镜像在构建时也会安装这些库,其中个别库安装失败不会中止构建,`sgl-kernel` 则不在镜像中;容器内缺少的库按仓库声明的版本另行安装。 +- 计时方式见 [benchmark 的计时方法](../timing.md)。 + +## 提交 PR { #pr } + +### 标题 { #pr-title } + +| 格式 | 例子 | +| --- | --- | +| `[Type] ` | `[Doc] Fix the install command in the README` | +| `[Type][Scope] ` | `[BugFix][Elementwise] Build the floored tiers at every tuned fold width` | +| `[Type][foundry][Scope] ` | `[Perf][foundry][Elementwise] Take the floored tier's reciprocal from one MUFU instruction` | + +- CI 检查 PR 标题的格式。`Type` 的取值定义在 [`.claude/conventions/types.sh`](https://github.com/tile-ai/TileOPs/blob/main/.claude/conventions/types.sh) 中。 +- `foundry` 标记 kernel 由 [TileFoundry](https://github.com/tile-ai/TileFoundry) 生成的 PR。 + +### 描述 { #pr-body } + +PR 描述按 [PR 模板](https://github.com/tile-ai/TileOPs/blob/main/.github/PULL_REQUEST_TEMPLATE.md)填写。改动涉及 `tests/` 时,描述中附上测试用例数量的变化: + +```bash +python scripts/test_node_delta.py --base upstream/main +``` + +### CI { #ci } + +- Draft PR 只运行 PR 标题检查与 manifest 统计,跳过下面的 CPU 检查与 GPU smoke 测试。PR 转为 **Ready for review** 后,这些检查才会运行。 +- CPU 检查包括 pre-commit、gitleaks、manifest 校验、actionlint、编译约定检查与打包检查;改动涉及 benchmark 时,另含 benchmark 约定测试。 +- GPU smoke 测试在 pre-commit、gitleaks 与 actionlint 通过后运行,测试范围由改动的文件决定。没有改动 Python、manifest 或原生代码的 PR 跳过 GPU smoke。 + +## 常见问题 { #faq } + +### 本地安装在重新解析 CUDA 或 TileLang 时失败 { #faq-rebuild } + +```bash +PIP_NO_BUILD_ISOLATION=1 pip install -e '.[dev]' -c constraints.txt +``` + +本机已安装 CUDA 与 TileLang 时,关闭构建隔离,构建过程直接使用已安装的版本。 + +### 没有 SM90 GPU 时可运行的检查 { #faq-no-gpu } + +[运行测试](#tests)一节中不需要 GPU 的三项检查。需要 GPU 的测试由 PR 上的 CI 运行。 diff --git a/docs/user-guide/index.md b/docs/user-guide/index.md index 8c82f1d5..bb973373 100644 --- a/docs/user-guide/index.md +++ b/docs/user-guide/index.md @@ -1,10 +1,30 @@ # User Guide -| Page | Contents | -| --- | --- | -| [Reading and writing the manifest](manifest/index.md) | how the system uses a spec, the concepts used to describe a spec, and how to write one | -| [How an op selects a kernel](dispatch/index.md) | how an op selects a kernel, how to add a kernel, and how a backend joins | -| [Adding a new op](../new-op.md) | the six steps from a spec to `status: implemented` | -| [Bringing an op into torch.compile](../torch-compile.md) | what an op looks like inside a compiled graph, and the conventions a caller follows | -| [How a benchmark is timed](../timing.md) | how the numbers on the Benchmarks pages are measured | -| [Adding a hardware backend](../backends.md) | serving the ops on one class of devices with your own kernels | +## Getting started + +- [Development guide](development.md) + Getting the code, setting up the dev image or a local environment, running the + tests, and opening a PR. + +## Specs and ops + +- [Reading and writing the manifest](manifest/index.md) + How the system uses a spec, the concepts a spec is described in, and how to + write one. +- [Adding a new op](../new-op.md) + The six steps from a spec to `status: implemented`. + +## Kernels and hardware + +- [How an op selects a kernel](dispatch/index.md) + How an op selects a kernel, how to add a kernel, and how a backend joins. +- [Adding a hardware backend](../backends.md) + Serving the ops on one class of devices with your own kernels. + +## Integration and performance + +- [Bringing an op into torch.compile](../torch-compile.md) + What an op looks like inside a compiled graph, and the conventions a caller + follows. +- [How a benchmark is timed](../timing.md) + How the numbers on the Benchmarks pages are measured. diff --git a/docs/user-guide/index.zh.md b/docs/user-guide/index.zh.md index ac2a3fd6..3e18c353 100644 --- a/docs/user-guide/index.zh.md +++ b/docs/user-guide/index.zh.md @@ -1,10 +1,27 @@ -# 使用指南 - -| 文档 | 内容 | -| --- | --- | -| [读写 manifest](manifest/index.md) | 系统如何使用 spec、描述 spec 所用的概念,以及如何写一份 spec | -| [op 如何选择 kernel](dispatch/index.md) | op 如何选中 kernel、如何新增 kernel,以及 backend 如何接入 | -| [添加新 op](../new-op.md) | 从一份 spec 到 `status: implemented` 的六步 | -| [接入 torch.compile](../torch-compile.md) | op 在编译图中的形态,以及调用时的约定 | -| [benchmark 的计时方法](../timing.md) | Benchmarks 页上的数字如何测得 | -| [接入新硬件 backend](../backends.md) | 用自己的 kernel 接管某一类设备上的 op | +# 用户指南 + +## 开始开发 + +- [开发指南](development.md) + 获取代码,搭建 dev 镜像或本地环境,运行测试,提交 PR。 + +## spec 与 op + +- [读写 manifest](manifest/index.md) + 系统如何使用 spec,描述 spec 所用的概念,以及 spec 的写法。 +- [添加新 op](../new-op.md) + 从一份 spec 到 `status: implemented` 的六个步骤。 + +## kernel 与硬件 + +- [op 如何选择 kernel](dispatch/index.md) + op 选中 kernel 的过程,新增 kernel 的方法,以及 backend 的接入方式。 +- [接入新硬件 backend](../backends.md) + 在某一类设备上,用自己的 kernel 实现 op。 + +## 集成与性能 + +- [接入 torch.compile](../torch-compile.md) + op 在编译图中的形态,以及调用时的约定。 +- [benchmark 的计时方法](../timing.md) + 性能数据页上的数字如何测得。 diff --git a/hooks.py b/hooks.py index 285606dd..9b7b7ded 100644 --- a/hooks.py +++ b/hooks.py @@ -12,6 +12,10 @@ a translation still to come rather than a broken one. * A blog post's first paragraph is its subtitle; `on_page_content` marks it for extra.css, so the post's Markdown carries no styling. +* A section index lists its pages as a plain Markdown list; `on_page_content` + marks those lists for extra.css, which draws each item as a card. +* A page merged into another leaves its old URL behind; `on_post_build` writes + a redirect there, so published links keep working. """ from __future__ import annotations @@ -71,9 +75,9 @@ def on_page_markdown(markdown, page, config, files): # not produce is left out; one it produced that is not listed here is appended. _BENCH_ORDER = [ "index.md", "reading.md", - "elementwise.md", "reduction.md", "normalization.md", "quantization.md", - "gemm.md", "conv-pool.md", "moe.md", "sampling.md", "rope.md", - "attention.md", "linear-attention.md", "ssm.md", "other.md", + "elementwise.md", "rope.md", "reduction.md", "normalization.md", + "conv-pool.md", "gemm.md", "quantization.md", "attention.md", "moe.md", + "sampling.md", "linear-attention.md", "ssm.md", "other.md", ] @@ -95,11 +99,59 @@ def on_config(config): _FIRST_PARAGRAPH = re.compile(r"(\s*)

") +# Section indexes whose page lists render as cards. Each item is a link followed +# by a one-line description on the next line. +_CARD_INDEXES = { + "user-guide/index.md", "user-guide/index.zh.md", "design/index.md", +} + def on_page_content(html, page, config, files): - """Mark a blog post's subtitle.""" + """Mark a blog post's subtitle, and the page lists on a section index.""" src = page.file.src_path.replace("\\", "/") name = src.split("/")[-1] + if src in _CARD_INDEXES: + return html.replace("

    ", '
      ') if not src.startswith("blog/") or name.startswith("index."): return html return _FIRST_PARAGRAPH.sub(r'\1

      ', html, count=1) + + +# Old page URL -> where its content lives now, relative to the old URL. The +# redirect is written into each locale's build. +_REDIRECTS = { + "api/topk/": "../sampling/#top-k-selection", +} + +_REDIRECT_PAGE = ( + '' + '' + 'Redirecting' + '{to}\n' +) + + +def _locale_dirs(config): + """The site root, then one subdirectory per other locale i18n builds. + + This hook runs before i18n builds the other locales, so their directories + are named from the plugin's config rather than found on disk. + """ + dirs = [""] + i18n = config["plugins"].get("i18n") + for lang in getattr(i18n, "config", {}).get("languages", []): + if lang.build and not lang.default: + dirs.append(lang.locale) + return dirs + + +def on_post_build(config): + """Write a redirect page at each old URL `_REDIRECTS` lists, per locale.""" + for locale in _locale_dirs(config): + for old, to in _REDIRECTS.items(): + path = os.path.join(config["site_dir"], locale, old, "index.html") + if os.path.exists(path): + continue + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "w", encoding="utf-8") as f: + f.write(_REDIRECT_PAGE.format(to=to)) diff --git a/mkdocs.yml b/mkdocs.yml index 17cae28a..73f60e20 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -32,7 +32,10 @@ nav: - Blog: - blog/index.md - User Guide: + # In the order the index groups them: getting started, specs and ops, + # kernels and hardware, integration and performance. - user-guide/index.md + - Development Guide: user-guide/development.md - Reading and Writing Specs: - user-guide/manifest/index.md - Concepts: user-guide/manifest/concepts.md @@ -40,39 +43,40 @@ nav: - Extensions: user-guide/manifest/extensions.md - Calls and Validation: user-guide/manifest/calls.md - Examples: user-guide/manifest/examples.md + - Adding an Op: new-op.md - Kernel Dispatch: - user-guide/dispatch/index.md - Adding a Kernel: user-guide/dispatch/writing.md - Backend Integration: user-guide/dispatch/backends.md - - Adding an Op: new-op.md - - Timing Benchmarks: timing.md - - Using torch.compile: torch-compile.md - Adding a Backend: backends.md - # In `tileops._FAMILIES` order — simple to composite: the pointwise transforms, - # then the axis reductions and the normalizations built on them, then the matmul - # and the expert routing over it, then the windowed and spectral transforms, then - # the sequence-model kernels built on all of the above. Pages are not one per - # family. The FP8 lightning indexer is a section of the Attention page, where its - # single op is used, and the sequence-modeling family is two pages, mHC and Engram, - # being two unrelated algorithms. Top-k keeps its page before GEMM though its op is - # exported from `tileops.attention`. Trace comes last — a tool rather than an op. + - Using torch.compile: torch-compile.md + - Timing Benchmarks: timing.md + # In the Benchmarks pages' order: the pointwise transforms with the positional + # rotation beside them, the axis reductions and the normalizations built on + # them, the windowed kernels, the matmul and the quantization around it, then + # attention, the expert routing after it and sampling at the end of a decode + # step, then the sequence-mixing kernels. The pages Benchmarks folds into Other + # follow: FFT, mHC and Engram. Top-k shares the Sampling page, though its op is + # exported from `tileops.attention` and its benchmark sits on Attention. Pages are not one per family. The FP8 + # lightning indexer is a section of the Attention page, where its single op is + # used, and the sequence-modeling family is two pages, mHC and Engram, being two + # unrelated algorithms. Trace comes last — a tool rather than an op. - API Reference: - api/index.md - Elementwise: api/elementwise.md + - RoPE: api/rope.md - Reduction: api/reduction.md - Normalization: api/normalization.md - - Quantization: api/quantization.md - - Top-k: api/topk.md - - GEMM: api/linear-algebra.md - Pooling: api/pool.md - Convolution: api/convolution.md - - FFT: api/fft.md - - MoE: api/moe.md - - Sampling: api/sampling.md - - RoPE: api/rope.md + - GEMM: api/linear-algebra.md + - Quantization & Dequantization: api/quantization.md - Attention: api/attention.md + - MoE: api/moe.md + - Top-k & Sampling: api/sampling.md - Linear Attention: api/linear-attention.md - Mamba: api/mamba.md + - FFT: api/fft.md - mHC: api/mhc.md - Engram: api/engram.md - Trace: api/trace.md @@ -85,16 +89,18 @@ nav: - performance-guides/memory-bound/index.md - Global Memory Access: performance-guides/memory-bound/global-memory-access.md - Shared Memory Access: performance-guides/memory-bound/shared-memory-access.md + # In the order the index groups them: the system's structure, the spec, building + # an op against it, then how an op is accepted. - Design: - design/index.md - Architecture: design/architecture.md + - Layer Boundaries: design/layer-boundaries.md - Manifest: design/manifest.md - Op Interfaces: design/ops-design.md - Op Interface Reference: design/ops-design-reference.md - Slot Rules: design/op-slot-rules.md - - Roofline: design/roofline.md - Testing: design/testing.md - - Layer Boundaries: design/layer-boundaries.md + - Roofline: design/roofline.md markdown_extensions: # Docstrings and design docs carry formulas. `generic: true` emits the plain @@ -164,7 +170,8 @@ plugins: Home: 首页 Blog: 博客 Design: 设计文档 - User Guide: 使用指南 + User Guide: 用户指南 + Development Guide: 开发指南 Reading and Writing Specs: 读写 manifest Concepts: 概念 Spec Fields: 写一个 spec @@ -181,6 +188,7 @@ plugins: Architecture: 架构 Manifest: manifest 规范 Op Interfaces: op 接口 + Op Interface Reference: op 接口参考 Slot Rules: Slot 规则 Roofline: Roofline Testing: 测试 @@ -191,6 +199,7 @@ plugins: Normalization: Normalization GEMM: GEMM Quantization: Quantization + Quantization & Dequantization: Quantization & Dequantization Attention: Attention RoPE: RoPE Convolution: Convolution @@ -202,7 +211,7 @@ plugins: MoE: MoE Sampling: Sampling Engram: Engram - Top-k: Top-k + Top-k & Sampling: Top-k & Sampling Trace: Trace Conv & Pool: Conv & Pool SSM: SSM diff --git a/scripts/gen_bench_pages.py b/scripts/gen_bench_pages.py index 8f85659d..67018301 100644 --- a/scripts/gen_bench_pages.py +++ b/scripts/gen_bench_pages.py @@ -2,8 +2,8 @@ """Render the Benchmarks section from a nightly benchmark XML snapshot. Output is one overview page, one page explaining the numbers, and the data -pages of `DATA_PAGES`, in the order the API Reference nav lists the same -families. `hooks.py` puts them into the site nav in that order. +pages of `DATA_PAGES`, in the order that list gives. `hooks.py` puts them into +the site nav in that order. These pages answer one question per workload: **how does TileOPs compare to the fastest other implementation of the same op on that workload?** @@ -96,7 +96,7 @@ def tier_of(tag: str) -> str: "ssm": "SSM", "scan": "Scan", "normalization": "Normalization", "moe": "MoE", "linear_algebra": "GEMM", "reduction": "Reduction", "elementwise": "Elementwise", "convolution": "Convolution", "pool": "Pooling", - "quantization": "Quantization", "sampling": "Sampling", "positional": "RoPE", + "quantization": "Quantization & Dequantization", "sampling": "Sampling", "positional": "RoPE", "fft": "FFT", "mhc": "mHC", "engram": "Engram", "topk": "Top-k", "other": "Other", } @@ -142,24 +142,24 @@ def api_op_order(api_dir: str = API_DIR, print(f"warning: no op order read from {api_dir}; every Benchmarks page " f"ranks its ops by verdict instead", file=sys.stderr) return order -# (slug, page title, families in display order), in the order the API Reference -# nav lists the same families — pointwise, then the reductions and the -# normalizations built on them, then quantization, then the matrix multiply and -# the expert routing over it, then the sampling, then the positional rotation, -# then the sequence-mixing kernels built on all of the above. A page is one -# family except where too few ops carry one: `Conv & Pool` is two, `Other` the -# rest. +# (slug, page title, families in display order). Pointwise first, with the +# positional rotation beside it, then the reductions and the normalizations +# built on them, the windowed kernels, the matrix multiply and the quantization +# around it, then attention and the expert routing that follows it in a +# transformer layer, sampling at the end of a decode step, then the +# sequence-mixing kernels. A page is one family except where too few ops carry +# one: `Conv & Pool` is two, `Other` the rest. DATA_PAGES = [ ("elementwise", "Elementwise", ["elementwise"]), + ("rope", "RoPE", ["positional"]), ("reduction", "Reduction", ["reduction"]), ("normalization", "Normalization", ["normalization"]), - ("quantization", "Quantization", ["quantization"]), - ("gemm", "GEMM", ["linear_algebra"]), ("conv-pool", "Conv & Pool", ["pool", "convolution"]), + ("gemm", "GEMM", ["linear_algebra"]), + ("quantization", "Quantization & Dequantization", ["quantization"]), + ("attention", "Attention", ["attention"]), ("moe", "MoE", ["moe"]), ("sampling", "Sampling", ["sampling"]), - ("rope", "RoPE", ["positional"]), - ("attention", "Attention", ["attention"]), ("linear-attention", "Linear Attention", ["linear_attention"]), ("ssm", "SSM", ["ssm"]), ("other", "Other", ["topk", "fft", "mhc", "engram", "scan", "other"]),