COURSE / SOURCE

merge_dram.py

All lessons
Source filecode/day49-roofline/merge_dram.py

This is the source used by the lesson and its recorded evidence. Compile commands and expected output live in the directory README.

#!/usr/bin/env python3
"""Put the compulsory byte count next to the measured one.

`./roofline --json` reports what each kernel asks global memory for, worked
from the kernel source. Nsight Compute reports what actually crossed each
level of the memory hierarchy, measured in hardware. Neither number is the
other one, and day 30 shipped a roofline that used the first where it needed
the second: its naive matmul read 1008.5 percent of the same card's copy
ceiling, which no memory system can do.

    ./roofline --json > run.jsonl
    sudo ncu --csv --clock-control base \\
        --metrics dram__bytes.sum,lts__t_bytes.sum,l1tex__t_bytes.sum \\
        ./roofline --profile > profile/day49-bytes.csv
    python3 merge_dram.py --json run.jsonl \\
        --ncu profile/day49-bytes.csv

The `sudo` is not decoration. A stock driver ships RmProfilingAdminOnly = 1,
so a plain `ncu` returns ERR_NVGPUCTRPERM. This script needs no GPU and no
profiler of its own, which is the point: a reader on Colab runs it against
the CSV shipped in profile/ and gets the same table.

Standard library only. Exit 0 ok, 1 a kernel the profiler never saw, 2 usage.
"""

import argparse
import csv
import json
import sys

# Written exactly as Nsight Compute writes them, because a reader who
# searches for one of these strings should land on this file and on the
# lesson. dram__bytes.sum is the one the roofline uses: sectors transferred
# multiplied by 32, "not derived from any algorithmic property of your
# kernel, but measured in HW", including write-backs, re-reads and sectors
# wasted by uncoalesced access. The quote is an NVIDIA engineer's, at
# https://forums.developer.nvidia.com/t/307220 (the forum redirects the
# short form to the full topic)
# (checked 2026-08-30).
LEVELS = [
    ("l1", "l1tex__t_bytes.sum"),
    ("l2", "lts__t_bytes.sum"),
    ("dram", "dram__bytes.sum"),
]


def read_rows(path):
    """One dict per line of the program's --json output."""
    rows = []
    with open(path, encoding="utf-8") as handle:
        for lineno, line in enumerate(handle, 1):
            line = line.strip()
            if not line.startswith("{"):
                continue
            try:
                rows.append(json.loads(line))
            except ValueError as err:
                sys.exit(f"{path}:{lineno}: not valid JSON: {err}")
    return rows


def read_metrics(path):
    """{kernel symbol: {metric name: bytes}} from an ncu --csv export.

    ncu emits one row per metric per launch and demangles the kernel into a
    full signature, so the join key is everything before the first bracket.
    --csv implies --print-units base, so the values are already in bytes and
    nothing here rescales them. Grouping separators are stripped because
    some locales emit them and a silent ValueError here would be a missing
    row rather than an error.
    """
    seen = {}
    with open(path, encoding="utf-8", newline="") as handle:
        # ncu prints warnings before the header, so skip to the line that
        # starts the table rather than trusting line one.
        lines = [ln for ln in handle if ln.lstrip().startswith('"')]
    if not lines:
        sys.exit(f"{path}: no quoted CSV rows; is this an ncu --csv export?")
    reader = csv.DictReader(lines)
    needed = {"Kernel Name", "Metric Name", "Metric Value"}
    missing = needed - set(reader.fieldnames or [])
    if missing:
        sys.exit(f"{path}: missing column(s) {sorted(missing)}")
    for row in reader:
        symbol = row["Kernel Name"].split("(")[0].strip()
        value = row["Metric Value"].replace(",", "").strip()
        try:
            seen.setdefault(symbol, {})[row["Metric Name"]] = float(value)
        except ValueError:
            continue
    return seen


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--json", required=True, help="./roofline --json")
    parser.add_argument("--ncu", required=True, help="ncu --csv export")
    args = parser.parse_args()

    rows = read_rows(args.json)
    metrics = read_metrics(args.ncu)

    ceilings = {r["variant"]: r for r in rows if "ceiling" in r["variant"]}
    ridge = next((r for r in rows if r["variant"] == "ridge-point"), None)
    if "ceiling-copy" not in ceilings or ridge is None:
        sys.exit(f"{args.json}: no ceiling-copy and ridge-point rows; this "
                 f"is not a ./roofline --json run")
    copy_gbs = ceilings["ceiling-copy"]["value"]

    kernels = [r for r in rows if "symbol" in r and "bytes" in r]
    if not kernels:
        sys.exit(f"{args.json}: no kernel rows with a byte count")

    print(f"copy ceiling {copy_gbs:.1f} GB/s, "
          f"ridge point {ridge['value']:.2f} FLOP/byte")
    print()
    head = ("kernel", "ask MB", "L1 MB", "L2 MB", "DRAM MB", "DRAM/ask",
            "%ceil")
    print("%-22s %8s %8s %8s %8s %8s %6s" % head)
    print("%-22s %8s %8s %8s %8s %8s %6s" %
          ("-" * 22, "-" * 8, "-" * 8, "-" * 8, "-" * 8, "-" * 8, "-" * 6))

    unseen = []
    for row in kernels:
        found = metrics.get(row["symbol"], {})
        if LEVELS[-1][1] not in found:
            unseen.append(row["symbol"])
        cells = []
        for _, metric in LEVELS:
            value = found.get(metric)
            cells.append("n/a" if value is None else "%8.1f" % (value / 1e6))
        dram = found.get("dram__bytes.sum")
        ask = row["bytes"]
        seconds = row["ms"] * 1e-3
        # Four decimals, not two. The interesting rows are the ones where
        # the caches absorbed almost everything, and at two decimals a
        # matmul that moved a three hundredth of what it asked for prints
        # as 0.00 and looks like a bug rather than the answer.
        ratio = "n/a" if dram is None else "%8.4f" % (dram / ask)
        pct = "n/a" if dram is None else "%6.1f" % (
            100.0 * (dram / seconds / 1e9) / copy_gbs)
        print("%-22s %8.1f %8s %8s %8s %8s %6s" %
              (row["variant"], ask / 1e6, cells[0], cells[1], cells[2],
               ratio, pct))

    print()
    print("ask MB is what the algorithm asks global memory for, worked from")
    print("the kernel source. The other three are measured. %ceil is the")
    print("DRAM column over elapsed time, against the copy ceiling, so it")
    print("is the number day 30's table should have printed.")

    if unseen:
        print(f"\nno dram__bytes.sum for: {', '.join(unseen)}",
              file=sys.stderr)
        print("the profiler never saw those launches, so their rows are "
              "empty rather than wrong", file=sys.stderr)
        return 1
    return 0


def demo():
    """The smallest thing that fails if the join breaks.

    Runs the parser over two literal fixtures with a known answer, so a
    change to either format shows up here rather than in a published table.
    """
    import os
    import tempfile

    csv_text = (
        '==PROF== disconnected\n'
        '"ID","Kernel Name","Metric Name","Metric Unit","Metric Value"\n'
        '"0","vectorAdd(float const *, float *, unsigned long)",'
        '"dram__bytes.sum","byte","201326592.0"\n'
        '"1","matmulNaive(float const *, float *, unsigned long)",'
        '"dram__bytes.sum","byte","3145728.0"\n'
    )
    with tempfile.TemporaryDirectory() as tmp:
        path = os.path.join(tmp, "bytes.csv")
        with open(path, "w", encoding="utf-8") as handle:
            handle.write(csv_text)
        got = read_metrics(path)
    want = {
        "vectorAdd": {"dram__bytes.sum": 201326592.0},
        "matmulNaive": {"dram__bytes.sum": 3145728.0},
    }
    if got != want:
        raise SystemExit(f"read_metrics: got {got}, want {want}")
    print("merge_dram self-check: ok")


if __name__ == "__main__":
    if len(sys.argv) > 1 and sys.argv[1] == "--self-check":
        demo()
    else:
        sys.exit(main())