diff --git a/GLOSSARY.md b/GLOSSARY.md index 4d21b7c..8497ed5 100644 --- a/GLOSSARY.md +++ b/GLOSSARY.md @@ -8,7 +8,7 @@ This file is generated from `gxray/glossary.py`. Edit that and run `just build-g ## Index -[DejaGnu](#dejagnu) | [GENERIC](#generic) | [GIMPLE](#gimple) | [IRA](#ira) | [LRA](#lra) | [RTL](#rtl) | [RTX](#rtx) | [SSA](#ssa) | [SSA name](#ssa-name) | [TODO flags](#todo-flags) | [allocno](#allocno) | [alternative](#alternative) | [assembler directive](#assembler-directive) | [back end](#back-end) | [basic block](#basic-block) | [blue paint](#blue-paint) | [bootstrap](#bootstrap) | [bubbling](#bubbling) | [build config](#build-config) | [cc1](#cc1) | [checking build](#checking-build) | [collect2](#collect2) | [compiler table](#compiler-table) | [constraint](#constraint) | [control flow graph](#control-flow-graph) | [cross compiler](#cross-compiler) | [current function](#current-function) | [debug counter](#debug-counter) | [default definition](#default-definition) | [define_insn](#define_insn) | [definition](#definition) | [directive](#directive) | [dominance](#dominance) | [driver](#driver) | [dump file](#dump-file) | [edge](#edge) | [effective target](#effective-target) | [excess errors](#excess-errors) | [expand](#expand) | [final](#final) | [front end](#front-end) | [garbage collector](#garbage-collector) | [gate](#gate) | [generated file](#generated-file) | [gengtype](#gengtype) | [gimplification](#gimplification) | [hard register](#hard-register) | [immediate dominator](#immediate-dominator) | [include guard](#include-guard) | [inferior call](#inferior-call) | [insn](#insn) | [interference](#interference) | [line marker](#line-marker) | [live range](#live-range) | [loop](#loop) | [machine description](#machine-description) | [machine mode](#machine-mode) | [middle end](#middle-end) | [mode iterator](#mode-iterator) | [optimization level](#optimization-level) | [out of SSA](#out-of-ssa) | [out of tree build](#out-of-tree-build) | [output template](#output-template) | [param](#param) | [pass](#pass) | [pass manager](#pass-manager) | [pass positioning](#pass-positioning) | [phi node](#phi-node) | [plugin](#plugin) | [plugin ABI](#plugin-abi) | [plugin event](#plugin-event) | [poly_int](#poly_int) | [port](#port) | [preprocessor](#preprocessor) | [pretty printer](#pretty-printer) | [pseudo register](#pseudo-register) | [pseudo-event](#pseudo-event) | [register allocation](#register-allocation) | [register class](#register-class) | [register pressure](#register-pressure) | [section](#section) | [spec](#spec) | [spec function](#spec-function) | [specs file](#specs-file) | [spill](#spill) | [stage comparison](#stage-comparison) | [stamp file](#stamp-file) | [sum file](#sum-file) | [target hook](#target-hook) | [target triple](#target-triple) | [temporary](#temporary) | [three address form](#three-address-form) | [token](#token) | [token pasting](#token-pasting) | [torture options](#torture-options) | [translation unit](#translation-unit) | [tree](#tree) | [use](#use) | [wide_int](#wide_int) +[DejaGnu](#dejagnu) | [GENERIC](#generic) | [GIMPLE](#gimple) | [IRA](#ira) | [LRA](#lra) | [RTL](#rtl) | [RTX](#rtx) | [SSA](#ssa) | [SSA name](#ssa-name) | [TODO flags](#todo-flags) | [allocno](#allocno) | [alternative](#alternative) | [assembler directive](#assembler-directive) | [back end](#back-end) | [basic block](#basic-block) | [blue paint](#blue-paint) | [bootstrap](#bootstrap) | [bubbling](#bubbling) | [build config](#build-config) | [cc1](#cc1) | [checking build](#checking-build) | [collect2](#collect2) | [compiler table](#compiler-table) | [constraint](#constraint) | [control flow graph](#control-flow-graph) | [cross compiler](#cross-compiler) | [current function](#current-function) | [debug counter](#debug-counter) | [default definition](#default-definition) | [define_insn](#define_insn) | [definition](#definition) | [diagnostic](#diagnostic) | [directive](#directive) | [dominance](#dominance) | [driver](#driver) | [dump file](#dump-file) | [edge](#edge) | [effective target](#effective-target) | [error recovery](#error-recovery) | [excess errors](#excess-errors) | [expand](#expand) | [final](#final) | [fix-it hint](#fix-it-hint) | [front end](#front-end) | [garbage collector](#garbage-collector) | [gate](#gate) | [generated file](#generated-file) | [gengtype](#gengtype) | [gimplification](#gimplification) | [hard register](#hard-register) | [immediate dominator](#immediate-dominator) | [include guard](#include-guard) | [inferior call](#inferior-call) | [insn](#insn) | [interference](#interference) | [line marker](#line-marker) | [live range](#live-range) | [lookahead](#lookahead) | [loop](#loop) | [machine description](#machine-description) | [machine mode](#machine-mode) | [middle end](#middle-end) | [mode iterator](#mode-iterator) | [optimization level](#optimization-level) | [out of SSA](#out-of-ssa) | [out of tree build](#out-of-tree-build) | [output template](#output-template) | [param](#param) | [parser](#parser) | [pass](#pass) | [pass manager](#pass-manager) | [pass positioning](#pass-positioning) | [phi node](#phi-node) | [plugin](#plugin) | [plugin ABI](#plugin-abi) | [plugin event](#plugin-event) | [poly_int](#poly_int) | [port](#port) | [preprocessor](#preprocessor) | [pretty printer](#pretty-printer) | [pseudo register](#pseudo-register) | [pseudo-event](#pseudo-event) | [register allocation](#register-allocation) | [register class](#register-class) | [register pressure](#register-pressure) | [section](#section) | [spec](#spec) | [spec function](#spec-function) | [specs file](#specs-file) | [spill](#spill) | [stage comparison](#stage-comparison) | [stamp file](#stamp-file) | [sum file](#sum-file) | [target hook](#target-hook) | [target triple](#target-triple) | [temporary](#temporary) | [three address form](#three-address-form) | [token](#token) | [token pasting](#token-pasting) | [torture options](#torture-options) | [translation unit](#translation-unit) | [tree](#tree) | [typedef name](#typedef-name) | [use](#use) | [wide_int](#wide_int) ## Reading the source @@ -282,6 +282,58 @@ While a macro is being expanded, its name is disabled; any occurrence of it in t Also written `NO_EXPAND`, painted blue, self-reference. Taught in F02. See also [preprocessor](#preprocessor), [token](#token). In the source: [`libcpp/macro.cc:1590@releases/gcc-16.2.0`](https://github.com/gcc-mirror/gcc/blob/releases/gcc-16.2.0/libcpp/macro.cc#L1590). +## In the parser + +The program that reads C, and the four token slots everything surprising about it comes out of. F03 is the lesson. + +### parser + +**The recursive descent code that turns a token stream into trees. It can see four tokens.** + +GCC's C parser is written by hand rather than generated, and all of its memory of your program is a `c_parser` struct: four token slots, a few flags, and the symbol table it shares with the rest of the front end. There is no dump flag for it, because it has no output of its own to print. What it produces is GENERIC, and what you can watch it do is complain. Nearly everything surprising about a C error message follows from the size of that buffer and from the fact that the symbol table has to answer a question before the parse can continue. + +Also written `c_parser`, recursive descent, `cc1`. Taught in F03. See also [lookahead](#lookahead), [typedef name](#typedef-name), [GENERIC](#generic). In the source: [`gcc/c/c-parser.cc:191@releases/gcc-16.2.0`](https://github.com/gcc-mirror/gcc/blob/releases/gcc-16.2.0/gcc/c/c-parser.cc#L191). + +### lookahead + +**How far ahead the parser can look before deciding what it is reading. In C, four tokens.** + +`c_parser_peek_token` gives the next one, `c_parser_peek_2nd_token` the one after it, and `c_parser_peek_nth_token` reaches as far as the fourth. The buffer behind all three is `c_token tokens_buf[4]` and nothing widens it. Most of the peeking in the C parser is one token deep, and the deepest constant peek in the whole file is there to recognise a version control conflict marker, which is not a C construct at all. When a grammar needs to see further than four, the parser does not get more; it commits, and then recovers. + +Also written peek, `tokens_buf`, LL(k). Taught in F03. See also [parser](#parser), [token](#token), [error recovery](#error-recovery). In the source: [`gcc/c/c-parser.cc:572@releases/gcc-16.2.0`](https://github.com/gcc-mirror/gcc/blob/releases/gcc-16.2.0/gcc/c/c-parser.cc#L572). + +### typedef name + +**An identifier that names a type, and the reason C cannot be parsed without a symbol table.** + +`A * b;` declares `b` as a pointer if `A` is a typedef name and multiplies two variables if it is not, and no amount of looking at tokens will tell you which. The parser asks the symbol table instead, and the answer is written into the token as `CPP_KEYWORD` or `CPP_NAME` the first time that token is looked at. That moment can come too early: a token peeked while one scope was open and used after it closed is carrying a stale answer, which is what `c_parser_maybe_reclassify_token` exists to undo. + +Also written lexer hack, `CPP_KEYWORD`, `c_parser_maybe_reclassify_token`. Taught in F03. See also [parser](#parser), [token](#token), [lookahead](#lookahead). In the source: [`gcc/c/c-parser.cc:2326@releases/gcc-16.2.0`](https://github.com/gcc-mirror/gcc/blob/releases/gcc-16.2.0/gcc/c/c-parser.cc#L2326). + +### diagnostic + +**One complaint, with a message, a severity, a place, and often a suggested repair.** + +A diagnostic is not a line of text. It is a structure with a primary location, any number of secondary ones, and any number of fix-it hints, and the text on your terminal is one rendering of it. `-fdiagnostics-format=sarif-stderr` is another, and it is the one to reach for when you want to read the structure rather than the prose. For the parser the message is finished by `c_parse_error`, which chooses one of thirteen endings from the type of the token the parser is looking at, which is why one missing semicolon can produce eight different sentences depending on what comes after it. + +Also written `-fdiagnostics-format`, SARIF, `c_parse_error`. Taught in F03. See also [fix-it hint](#fix-it-hint), [parser](#parser), [pretty printer](#pretty-printer). In the source: [`gcc/c-family/c-common.cc:7004@releases/gcc-16.2.0`](https://github.com/gcc-mirror/gcc/blob/releases/gcc-16.2.0/gcc/c-family/c-common.cc#L7004). + +### fix-it hint + +**A machine applicable edit hung off a diagnostic, saying insert or delete or replace this text here.** + +GCC does not suggest a repair for every mistake. For a missing token it suggests one only for the seven token types `get_missing_token_insertion_kind` knows, and the hint is what decides where the caret goes: two of the seven are inserted before the token that upset the parser and five after the token before it, and in the second case the caret moves back to the end of that previous token. That is why an error about a semicolon points at the line above the one you were reading. `-fdiagnostics-parseable-fixits` prints the hints in a form an editor can apply. + +Also written `fixit_hint`, `rich_location`, `-fdiagnostics-parseable-fixits`. Taught in F03. See also [diagnostic](#diagnostic), [parser](#parser). In the source: [`libcpp/include/rich-location.h:620@releases/gcc-16.2.0`](https://github.com/gcc-mirror/gcc/blob/releases/gcc-16.2.0/libcpp/include/rich-location.h#L620). + +### error recovery + +**What the parser does after an error so that it can carry on and find the next one.** + +A parser that stopped at the first mistake would make you compile a file once per typo, so after complaining it throws tokens away until it reaches one it can start again from, usually a semicolon or a closing brace at the right nesting depth. This is why three missing semicolons can come out as two errors rather than three, and why an error near the end of a file is sometimes a consequence of one near the top rather than a mistake of its own. The habit to build is to fix the first error and compile again. + +Also written resynchronise, `c_parser_skip_until_found`. Taught in F03. See also [parser](#parser), [diagnostic](#diagnostic), [lookahead](#lookahead). In the source: [`gcc/c/c-parser.cc:1353@releases/gcc-16.2.0`](https://github.com/gcc-mirror/gcc/blob/releases/gcc-16.2.0/gcc/c/c-parser.cc#L1353). + ## The four shapes a function takes The same function, written down four different ways on its way to assembly. T02, T03 and T07 are the lessons. diff --git a/README.md b/README.md index 84afd59..d2c772f 100644 --- a/README.md +++ b/README.md @@ -58,8 +58,9 @@ Lessons do not invent a fresh example each time either. There are three programs | B05 | [Sixty lines of C++, and you are inside the compiler](https://github.com/tamnd/gcc-internals/blob/main/lessons/b05-the-plugin/b05.ipynb) | GCC's plugin mechanism from the outside in: the three things a plugin has to have, the three ways it is refused, what an event actually is, a GIMPLE pass of your own inserted after ssa with its own dump file, switching one of GCC's passes off from outside and watching the assembly move, and why the thing you are writing against is not an API | M2 | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/tamnd/gcc-internals/blob/main/lessons/b05-the-plugin/b05.ipynb) | | F01 | [The driver is an interpreter](https://github.com/tamnd/gcc-internals/blob/main/lessons/f01-the-spec-language/f01.ipynb) | That `gcc -dumpspecs` prints a program, in a language with conditionals and function calls, which the driver interprets to decide what to run; how to read it; and four text files that change what your compiler does without patching or rebuilding it | M3 | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/tamnd/gcc-internals/blob/main/lessons/f01-the-spec-language/f01.ipynb) | | F02 | [The preprocessor is not a text editor](https://github.com/tamnd/gcc-internals/blob/main/lessons/f02-tokens-not-text/f02.ipynb) | That the preprocessor lexes your file into tokens before it does anything else, and that its printed output is a rendering of those tokens rather than the tokens themselves; the space GCC inserts that is in no input file; the four hundred macros you did not write; and why one #include of stdio.h opens thirty eight files | M3 | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/tamnd/gcc-internals/blob/main/lessons/f02-tokens-not-text/f02.ipynb) | +| F03 | [The C parser can see four tokens](https://github.com/tamnd/gcc-internals/blob/main/lessons/f03-four-tokens/f03.ipynb) | That GCC's C parser is hand written recursive descent whose whole memory is four token slots and the symbol table; why one missing semicolon produces eight different messages; why the caret is on the line above the mistake; why three mistakes come out as two errors; and why `A * b;` needs a symbol table to read | M3 | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/tamnd/gcc-internals/blob/main/lessons/f03-four-tokens/f03.ipynb) | -19 of 96 written. +20 of 96 written. This table is generated from the lessons, by the same command that builds them, so it cannot list a lesson that does not exist or miss one that does. T05 is the pilot and it is deliberately in the middle of Part I rather than at the start, because everything the course promises has to be true of a hard lesson before it is worth writing the easy ones. diff --git a/blueprints/BP-CPARSE.md b/blueprints/BP-CPARSE.md new file mode 100644 index 0000000..74243e5 --- /dev/null +++ b/blueprints/BP-CPARSE.md @@ -0,0 +1,628 @@ +# BP-CPARSE, the C parser + +**Status:** partial +**Applies to:** GCC 16.2.0 (tag `releases/gcc-16.2.0`) +**Target-dependent:** no +**Generated sections:** none +**Last verified:** 2026-09-06 against `releases/gcc-16.2.0` + +This document specifies the C front end's parser: the code in `gcc/c/c-parser.cc` and `gcc/c/c-parser.h` that turns the token stream libcpp produced into calls on the tree building interface in `gcc/c/c-decl.cc` and `gcc/c/c-typeck.cc`. The token, the parser state, the four slot lookahead buffer and the identifier classification are specified field by field. Lexing one token, peeking, consuming, the three disambiguations C cannot make without a symbol table, error reporting, fix-it insertion, caret placement and the four recovery routines are specified as algorithms. Semantic analysis, the tree building interface, attributes, OpenMP, OpenACC, Objective-C, transactional memory and the `__RTL` and `__GIMPLE` function body parsers are named and not specified, and each place that stops short says so. Nothing here is generated, because the parser keeps no tables in `.def` files: its grammar is control flow and its keyword set is a C enum. What exists instead is `gxray.cparse`, a reader for recorded diagnostics, with tests that compare its transcription of `get_missing_token_insertion_kind` and of the thirteen `c_parse_error` branches against the pinned tree, so a GCC that grows a case fails the build rather than making a paragraph quietly false. + +## 1. Purpose and scope + +The C parser is a hand written recursive descent parser. There is no grammar file, no generated table and no parser generator anywhere in the C front end. The file header at `gcc/c/c-parser.h:4@releases/gcc-16.2.0` records that the actions came from an older Bison parser and that the structure was influenced by the C++ parser, and the Bison grammar itself has been gone since GCC 4.1. + +The parser's contract with the code above it is one function. `c_parse_file` at `gcc/c/c-parser.cc:31269@releases/gcc-16.2.0` is called once per translation unit by the language hook `c_common_parse_file`, and returns when the token stream is exhausted. Its contract with the code below it is also one function: `c_lex_with_flags` in `gcc/c/c-lex.cc`, which wraps `cpp_get_token` and is called from exactly one place in the parser, `c_lex_one_token` at `gcc/c/c-parser.cc:336@releases/gcc-16.2.0`. Everything the parser will ever know about the program arrives through that one call. + +**What this document covers.** The `c_token` and `c_parser` records. The lookahead buffer and the assertions that bound it. How a preprocessing token becomes a parser token, including the symbol table lookup that decides whether an identifier names a type. The three constructs C cannot disambiguate from tokens alone, and how each is resolved. The top level parse loop. How an error is reported, where the caret goes, when a fix-it hint is offered and what offering one does to the caret. The error latch. The four skip routines that implement recovery. The conflict marker recognizer, which is the only thing in the front end that needs a fourth token of lookahead. + +**What it does not cover.** Semantic analysis: everything the parser calls in `c_decl.cc`, `c-typeck.cc`, `c-convert.cc` and `c-fold.cc`, which is where declarations become `tree` nodes and where types are checked. The GENERIC the parser builds, which is `BP-GENERIC`. Attribute parsing, both the GNU `__attribute__` syntax and the C23 `[[...]]` syntax, which reaches into `gcc/c-family/c-attribs.cc` and has its own lookahead machinery in `c_parser_check_balanced_raw_token_sequence`. The OpenMP, OpenACC and Objective-C parsers, which between them are more of `c-parser.cc` than C is. Transactional memory. The `__RTL` and `__GIMPLE` body parsers at `gcc/c/c-parser.cc:31308@releases/gcc-16.2.0`, which reopen the source file and parse it again by character. Pragma dispatch beyond the point at which a `CPP_PRAGMA` token interrupts the stream. Precompiled header loading, which happens before the first ordinary token is looked at. + +**Position in the pipeline.** Second. Its input is what `BP-CPP` describes and its output is what `BP-GENERIC` describes. + +**Inputs and outputs as properties.** None. The `PROP_*` flags describe IR held by the middle end, and the parser produces GENERIC, which predates them. Its input is a token stream and its output is a sequence of calls that leave declarations in the symbol table and function bodies on the statement list. + +## 2. Data structures + +### 2.1 The token + +`c_token` at `gcc/c/c-parser.h:53@releases/gcc-16.2.0`. It is not `cpp_token`. The preprocessor's token is described in section 2.1 of `BP-CPP`, and this is what the front end makes of one after string literal concatenation and after conversion of preprocessing tokens to tokens. + +| Field | Type | Meaning | +|---|---|---| +| `type` | `cpp_ttype` in 8 bits | which of the token kinds this is, from libcpp's table | +| `id_kind` | `c_id_kind` in 8 bits | for a `CPP_NAME`, what the symbol table said it was | +| `keyword` | `rid` in 8 bits | for a `CPP_KEYWORD`, which keyword, otherwise `RID_MAX` | +| `pragma_kind` | `pragma_kind` in 8 bits | for a `CPP_PRAGMA`, which pragma, otherwise `PRAGMA_NONE` | +| `location` | `location_t` | a key into the line map, not a line number | +| `value` | `tree` | the payload: an `IDENTIFIER_NODE`, a constant, or `NULL_TREE` | +| `flags` | `unsigned char` | libcpp's token flags, truncated from `unsigned short` | + +Two of those fields carry information libcpp did not have. `id_kind` is the answer to a symbol table query, described in section 3.2. `keyword` is set by looking the identifier up in the keyword table, which is why `CPP_KEYWORD` is a parser token type and not a preprocessor one: libcpp has no keywords, only identifiers. + +`flags` is one byte here and two in `cpp_token`, so the parser sees the low eight of libcpp's thirteen flags and loses the rest. The comment at `gcc/c/c-parser.cc:1122@releases/gcc-16.2.0` records the consequence: `c_parse_error` is passed a flag word of zero, so a token whose spelling depended on a flag, such as a digraph, is reported by its canonical spelling rather than the one you wrote. + +### 2.2 What an identifier turned out to be + +`c_id_kind` at `gcc/c/c-parser.h:38@releases/gcc-16.2.0`. Five values, and the first two are the whole of C. + +| Value | Meaning | +|---|---| +| `C_ID_ID` | an ordinary identifier | +| `C_ID_TYPENAME` | an identifier declared as a typedef name | +| `C_ID_CLASSNAME` | an Objective-C class name | +| `C_ID_ADDRSPACE` | a target defined address space qualifier | +| `C_ID_NONE` | the token is not an identifier | + +`C_ID_ADDRSPACE` is the one place a target influences parsing. The names come from the port's `ADDR_SPACE_KEYWORDS`, and `targetm.addr_space.diagnose_usage` is called at classification time, at `gcc/c/c-parser.cc:392@releases/gcc-16.2.0`. Nothing else in this document is target dependent. + +### 2.3 The parser + +`c_parser` at `gcc/c/c-parser.cc:191@releases/gcc-16.2.0`. There is one of these per translation unit, in the global `the_parser`, and it is the parser's entire memory of the program. Fields that belong to OpenMP, Objective-C or transactional memory are listed for completeness and are out of scope. + +| Field | Type | Meaning | +|---|---|---| +| `tokens` | `c_token *` | the lookahead window, normally `&tokens_buf[0]` | +| `tokens_buf` | `c_token[4]` | the buffer behind it | +| `tokens_avail` | `unsigned` | how many of the window are filled, 0 to 4 | +| `raw_tokens` | `vec *` | unclassified lookahead, for `[[` in Objective-C | +| `raw_tokens_used` | `unsigned` | how many of those have since been lexed properly | +| `error` | 1 bit | a syntax error is being recovered from | +| `in_pragma` | 1 bit | inside a pragma, so `CPP_PRAGMA_EOL` is not consumed automatically | +| `in_if_block` | 1 bit | parsing the outermost block of an `if` | +| `lex_joined_string` | 1 bit | libcpp should join adjacent string literals, for `#pragma pch_preprocess` | +| `translate_strings_p` | 1 bit | convert string literals to the execution character set | +| `seen_string_literal` | 1 bit | the last thing returned was a string literal | +| `last_token_location` | `location_t` | where the token most recently consumed was | +| `objc_*` | 4 bits total | Objective-C lexical context, out of scope | +| `in_transaction` | 4 bits | transactional memory, out of scope | +| `omp_*`, `in_omp_*` | pointers and a bit | OpenMP, out of scope | + +Three of those fields are the subject of most of this document. `tokens_avail` bounds the lookahead. `error` is the latch that stops one mistake becoming a hundred messages. `last_token_location` is the only thing that makes a fix-it hint for a missing token possible, because a hint has to be placed after the token before the one that upset the parser, and by then that token is gone. + +The struct is `GTY(())`, so it is garbage collector visible, and `tokens` is `GTY((skip))` because it points into `tokens_buf` rather than owning anything. The parser is collected at the end of the translation unit and not before; `ggc_collect` is called once per external declaration from the top level loop, and the parser survives it. + +### 2.4 The lookahead window + +The window is not a ring buffer and not a queue. `tokens` is a pointer that normally aims at `tokens_buf[0]`, and consuming a token shuffles the remaining entries down by one, at `gcc/c/c-parser.cc:969@releases/gcc-16.2.0`. When the parser is replaying a pre-lexed token vector, which happens for OpenMP attribute syntax, `tokens` points into that vector instead and consuming advances the pointer, and in that mode `tokens_avail` may exceed 4. + +While parsing C, `tokens_avail` is at most 4 and the reachable depth is exactly 4. The comment above the struct says two, and has said two since before the fourth slot was added. + +### 2.5 Operator precedence + +`c_parser_prec` at `gcc/c/c-parser.h:118@releases/gcc-16.2.0`. Ten levels and a dummy bottom, used by the binary expression parser to run an operator precedence loop rather than eleven mutually recursive functions. + +`PREC_NONE`, `PREC_LOGOR`, `PREC_LOGAND`, `PREC_BITOR`, `PREC_BITXOR`, `PREC_BITAND`, `PREC_EQ`, `PREC_REL`, `PREC_SHIFT`, `PREC_ADD`, `PREC_MULT`. + +### 2.6 How hard to try when guessing at a type name + +`c_lookahead_kind` at `gcc/c/c-parser.h:133@releases/gcc-16.2.0`. Three values, passed down to the predicate in section 3.4. + +| Value | Meaning | +|---|---| +| `cla_prefer_type` | an unknown identifier here is a type name | +| `cla_nonabstract_decl` | an unknown identifier is a type name if followed by an identifier or `*` | +| `cla_prefer_id` | never guess | + +## 3. Algorithms + +### 3.1 The top level + +```text +function parse_file (parser: Parser) + complexity: O(n) in tokens, amortised + + if peek(parser, 1).pragma_kind == PRAGMA_GCC_PCH_PREPROCESS + parse_pch_preprocess (parser) + else + no_more_pch () + + if peek(parser, 1).type == CPP_EOF + pedwarn ("ISO C forbids an empty translation unit") + return + + repeat + collect_garbage () + parse_external_declaration (parser) + until peek(parser, 1).type == CPP_EOF +``` + +`c_parser_translation_unit` at `gcc/c/c-parser.cc:2081@releases/gcc-16.2.0`. The obstack save and restore around the body of the loop are omitted: the parser allocates its scratch on `parser_obstack` and frees back to a mark after each external declaration, so the peak scratch is one declaration rather than one file. + +The loop has no error handling in it. An external declaration that fails to parse recovers inside `parse_external_declaration`, and the loop cannot tell the difference. + +### 3.2 Lexing one token + +```text +function lex_one (parser: Parser) -> Token + complexity: O(1) amortised, plus whatever libcpp does + + t = Token() + t.type = lex_with_flags (&t.value, &t.location, &t.flags) + t.id_kind = C_ID_NONE + t.keyword = RID_MAX + t.pragma_kind = PRAGMA_NONE + + if t.type == CPP_NAME + t.id_kind = C_ID_ID + rid = rid_code (t.value) + if rid is an address space keyword + diagnose_address_space_usage (rid, t.location) + t.id_kind = C_ID_ADDRSPACE + t.keyword = rid + return t + if rid != RID_MAX + t.type = CPP_KEYWORD + t.keyword = rid + return t + decl = lookup_name (t.value) + if decl != nothing and code(decl) == TYPE_DECL + t.id_kind = C_ID_TYPENAME + else + t.id_kind = C_ID_ID + else if t.type == CPP_PRAGMA + t.pragma_kind = integer_value (t.value) + t.value = nothing + + return t +``` + +`c_lex_one_token` at `gcc/c/c-parser.cc:336@releases/gcc-16.2.0`, with the Objective-C context sensitive keyword handling at `gcc/c/c-parser.cc:397@releases/gcc-16.2.0` omitted. + +The `lookup_name` call is the lexer hack, and its position in this function is the whole of what makes C hard to parse and the origin of every problem in sections 3.5 and 6.4. The classification is written into the token and the token is then held in the lookahead window, so the answer is the one that was true at the moment the token was first looked at, not the one that is true when the parser gets round to using it. + +`lookup_name` is not a hash table probe. It walks the scope chain in `gcc/c/c-decl.cc`, so classification cost is proportional to scope depth, and it has the side effect of marking the binding used for `-Wunused` purposes. + +### 3.3 Peeking and consuming + +```text +function peek (parser: Parser, n: integer) -> Token + complexity: O(1) + + assert n > 0 + if parser.tokens_avail >= n + return parser.tokens[n - 1] + assert parser.tokens_avail == n - 1 + parser.tokens[n - 1] = lex_one (parser) + parser.tokens_avail = n + return parser.tokens[n - 1] + +function consume (parser: Parser) + complexity: O(1) + + assert parser.tokens_avail >= 1 + assert parser.tokens[0].type != CPP_EOF + parser.last_token_location = parser.tokens[0].location + if parser.tokens is not &parser.tokens_buf[0] + parser.tokens = parser.tokens + 1 + else + for i in 0 .. parser.tokens_avail - 2 + parser.tokens[i] = parser.tokens[i + 1] + parser.tokens_avail = parser.tokens_avail - 1 + parser.seen_string_literal = false +``` + +`c_parser_peek_nth_token` at `gcc/c/c-parser.cc:572@releases/gcc-16.2.0` and `c_parser_consume_token` at `gcc/c/c-parser.cc:958@releases/gcc-16.2.0`. `c_parser_peek_token` and `c_parser_peek_2nd_token` are the same function specialised to n of 1 and 2. + +Two properties follow from the assertion in `peek`. Lookahead has to be requested in order: asking for the third token without having asked for the second is an internal compiler error, not a longer read. And the parser can never look past the end of the file, because `peek` at 2 asserts that the first token is not `CPP_EOF` and lexing past EOF would otherwise be possible. + +`consume` does not refill. The window is filled lazily, one token per peek, so a parser that never peeks past the first token never lexes past it either. + +### 3.4 What starts a declaration + +C's first ambiguity is that a statement and a declaration may both begin with an identifier. The parser resolves it with a predicate over up to two tokens. + +```text +function starts_declaration (parser: Parser, n: integer) -> boolean + complexity: O(1) + + t = peek (parser, n) + if t.type == CPP_NAME and peek(parser, n + 1).type == CPP_COLON + return false -- a label + if t.keyword == RID_STATIC_ASSERT and peek(parser, n + 1).type == CPP_OPEN_PAREN + return the balanced parenthesis run is followed by a semicolon + if starts_declspecs (t) or t.keyword == RID_STATIC_ASSERT + return true + return starts_typename (parser, cla_nonabstract_decl, n) + +function starts_typename (parser: Parser, la: LookaheadKind, n: integer) -> boolean + complexity: O(1), plus one symbol table walk in the guessing branch + + t = peek (parser, n) + if t.type == CPP_NAME + if t.id_kind in {C_ID_TYPENAME, C_ID_ADDRSPACE, C_ID_CLASSNAME} + return true + if t.type == CPP_KEYWORD + return keyword_starts_typename (t.keyword) + if la == cla_prefer_id + return false + if t.type != CPP_NAME or t.id_kind != C_ID_ID + return false + if la != cla_prefer_type + and peek(parser, n + 1).type not in {CPP_NAME, CPP_MULT} + return false + return lookup_name (t.value) == nothing +``` + +`c_parser_next_tokens_start_declaration` at `gcc/c/c-parser.cc:916@releases/gcc-16.2.0` and `c_parser_next_tokens_start_typename` at `gcc/c/c-parser.cc:696@releases/gcc-16.2.0`. + +The last three lines are a guess and are labelled as one in the source. An identifier that is not declared at all, followed by another identifier or by a `*`, is treated as a type name that the user got wrong. This is what turns `foo bar;` into `unknown type name 'foo'` rather than into a syntax error, and it is why the two shapes it tests for are exactly the two that a misspelled type would produce. + +The static assertion case is the only place in the C parser that reads an unbounded run of tokens ahead. It does so through the raw token vector rather than through the four slot window, which is what `raw_tokens` in section 2.3 is for. + +### 3.5 Reclassifying a stale token + +```text +function maybe_reclassify (parser: Parser) + complexity: O(1), plus one symbol table walk + + if peek(parser, 1).type != CPP_NAME + return + t = peek (parser, 1) + if t.id_kind not in {C_ID_ID, C_ID_TYPENAME} + return + decl = lookup_name (t.value) + t.id_kind = C_ID_ID + if decl != nothing and code(decl) == TYPE_DECL + t.id_kind = C_ID_TYPENAME +``` + +`c_parser_maybe_reclassify_token` at `gcc/c/c-parser.cc:2326@releases/gcc-16.2.0`. It is called from five places, all of them the point at which a construct that opened a scope has closed it again: after an `if` statement at `gcc/c/c-parser.cc:8907@releases/gcc-16.2.0`, after a `switch`, after a `while`, after a `for`, and after the declaration list of an old style function definition. + +The call is needed because a statement's body can be parsed while the enclosing `for` header's scope is still open, and the token after the body will already have been peeked and classified against that scope. The bug is PR67784 and the comment names it. Note what the function does not do: it does not re-peek, it overwrites `id_kind` in place, and it clears `C_ID_CLASSNAME` and `C_ID_ADDRSPACE` by not being called for them. + +### 3.6 Reporting an error + +```text +function report_at (parser: Parser, message: string, richloc: RichLocation) -> boolean + complexity: O(1) + + t = peek (parser, 1) + if parser.error + return false + parser.error = true + if message == nothing + return false + + if t.type in {CPP_LSHIFT, CPP_RSHIFT, CPP_EQ_EQ} + loc = peek_conflict_marker (parser, t.type) + if loc != nothing + error_at (loc, "version control conflict marker in file") + return true + + if parser.seen_string_literal and t.type == CPP_NAME + header = stdlib_header_for_string_macro (spelling (t.value)) + if header != nothing + attach_missing_header_hint (richloc, header) + + kind = t.type + if kind == CPP_KEYWORD + kind = CPP_NAME + parse_error (message, kind, t.value, 0, richloc) + return true + +function report (parser: Parser, message: string) -> boolean + complexity: O(1) + + t = peek (parser, 1) + if t.type != CPP_EOF + input_location = t.location + return report_at (parser, message, RichLocation(input_location)) +``` + +`c_parser_error_richloc` at `gcc/c/c-parser.cc:1073@releases/gcc-16.2.0` and `c_parser_error` at `gcc/c/c-parser.cc:1135@releases/gcc-16.2.0`. + +Three things in that are worth stating separately. The latch is the first two lines, and it is checked before anything else, so a second error while `parser.error` is set costs nothing and produces nothing. The keyword collapse near the end means `c_parse_error` never sees `CPP_KEYWORD`, which is why `expected ';' before 'while'` uses the identifier branch of section 3.8 rather than a keyword branch, and there is no keyword branch. And `report` guards the `input_location` update on the token not being `CPP_EOF`, at `gcc/c/c-parser.cc:1009@releases/gcc-16.2.0`, so an error at end of input is reported at wherever the parser last was rather than at a location past the end of the file. + +### 3.7 Requiring a token, and the caret swap + +```text +function require (parser: Parser, type: TokenType, message: string, + matching: Location, type_is_unique: boolean) -> boolean + complexity: O(1) + + if peek(parser, 1).type == type + consume (parser) + return true + + richloc = RichLocation (peek(parser, 1).location) + if not parser.error and type_is_unique + maybe_suggest_insertion (richloc, type, parser.last_token_location) + folded = false + if matching != unknown + folded = add_location_if_nearby (richloc, matching) + if report_at (parser, message, richloc) + if matching != unknown and not folded + inform (matching, "to match this %qs", symbol_for (type)) + return false + +function maybe_suggest_insertion (richloc: RichLocation, type: TokenType, + prev: Location) + complexity: O(1) + + kind = insertion_kind (type) + if kind == impossible + return + if kind == before_next + hint_at = immediately before richloc.primary + else + hint_at = immediately after prev + add_fixit_insert (richloc, hint_at, spelling (type)) + swap (richloc.primary, hint_at) +``` + +`c_parser_require` at `gcc/c/c-parser.cc:1279@releases/gcc-16.2.0`, `maybe_suggest_missing_token_insertion` at `gcc/c-family/c-common.cc:10049@releases/gcc-16.2.0` and `get_missing_token_insertion_kind` at `gcc/c-family/c-common.cc:9974@releases/gcc-16.2.0`. + +`insertion_kind` is a switch over seven token types and a default: + +| Token | Where the hint goes | +|---|---| +| `[` | before the token the parser is looking at | +| `(` | before the token the parser is looking at | +| `)` | after the token before it | +| `]` | after the token before it | +| `;` | after the token before it | +| `,` | after the token before it | +| `:` | after the token before it | + +Every other token type gets no hint, and the last line of `maybe_suggest_insertion` never runs for it, so the caret stays on the token that upset the parser. + +The swap is what puts the caret on the line above the mistake, and GCC's own comment at `gcc/c-family/c-common.cc:10012@releases/gcc-16.2.0` explains it with a diagram. The old primary location is not discarded: it becomes a secondary range on the same diagnostic, which is how an outside observer can tell a swapped diagnostic from an unswapped one without reading the source. Section 5.2 records that. + +`type_is_unique` is false when the caller could accept more than one token, which in practice means the message names two, as in `expected ',' or ';'`. Suggesting an insertion when two different tokens would do is not something a fix-it hint can express, so no hint is offered and no swap happens. + +### 3.8 Finishing the sentence + +`c_parse_error` at `gcc/c-family/c-common.cc:7004@releases/gcc-16.2.0` takes a complaint and a token type and returns a whole sentence. It is shared with the C++ front end. The branch is chosen by the token type alone, and there are thirteen. + +| Appended | Selected by | +|---|---| +| ` at end of input` | `CPP_EOF` | +| ` before %s'%c'` | the printable character constant types | +| ` before %s'\x%x'` | the same, when the value does not print | +| ` before user-defined character literal` | the `*_USERDEF` character types | +| ` before user-defined string literal` | the `*_USERDEF` string types | +| ` before string constant` | the string types | +| ` before numeric constant` | `CPP_NUMBER` | +| ` before %qE` | `CPP_NAME` | +| ` before %<#pragma%>` | `CPP_PRAGMA` | +| ` before end of line` | `CPP_PRAGMA_EOL` | +| ` before %` | `CPP_DECLTYPE` | +| ` before %<#embed%>` | `CPP_EMBED` | +| ` before %qs token` | everything else, by a range test | + +The last branch is where every punctuation mark in the language lands, which is why the word `token` appears on the end of some of these messages and not others. Nothing about the parser's state contributes: the same complaint before the same token type produces the same sentence wherever it came from. + +The printed forms of two branches collide. `%qE` on a one letter identifier and `%s'%c'` on a character constant both produce a single character in single quotes, so a reader who has only the message cannot tell which fired. Section 5.1 records that this is observable and that it is not recoverable from the text. + +### 3.9 Recovery + +Four routines, and which one a caller picks decides how much of your program is thrown away. + +```text +function skip_until_found (parser: Parser, type: TokenType, message: string, + matching: Location) + complexity: O(n) in tokens skipped + + if require (parser, type, message, matching) + return + depth = 0 + loop + t = peek (parser, 1) + if t.type == type and depth == 0 + consume (parser) + break + if t.type == CPP_EOF + return + if t.type == CPP_PRAGMA_EOL and parser.in_pragma + return + if t.type in {CPP_OPEN_BRACE, CPP_OPEN_PAREN, CPP_OPEN_SQUARE} + depth = depth + 1 + else if t.type in {CPP_CLOSE_BRACE, CPP_CLOSE_PAREN, CPP_CLOSE_SQUARE} + if depth == 0 + break + depth = depth - 1 + consume (parser) + parser.error = false + +function skip_to_end_of_block_or_statement (parser: Parser) + complexity: O(n) in tokens skipped + + depth = 0 + loop + t = peek (parser, 1) + if t.type == CPP_EOF + return + if t.type == CPP_PRAGMA_EOL and parser.in_pragma + return + if t.type == CPP_SEMICOLON and depth == 0 + consume (parser) + break + if t.type == CPP_CLOSE_BRACE and (depth == 0 or depth - 1 == 0) + consume (parser) + break + if t.type == CPP_OPEN_BRACE + depth = depth + 1 + else if t.type == CPP_CLOSE_BRACE + depth = depth - 1 + else if t.type == CPP_PRAGMA + consume the pragma up to and including its CPP_PRAGMA_EOL + continue + consume (parser) + parser.error = false +``` + +`c_parser_skip_until_found` at `gcc/c/c-parser.cc:1353@releases/gcc-16.2.0` and `c_parser_skip_to_end_of_block_or_statement` at `gcc/c/c-parser.cc:1594@releases/gcc-16.2.0`. The other two are `c_parser_skip_to_end_of_parameter` at `gcc/c/c-parser.cc:1426@releases/gcc-16.2.0`, which stops at a comma or a semicolon at depth zero without consuming it, and `c_parser_skip_to_pragma_eol` at `gcc/c/c-parser.cc:1507@releases/gcc-16.2.0`. + +All four clear `parser.error` on the way out, and that is the only way it gets cleared: there are eighteen assignments of `false` to it in the file and every one is a recovery point. Until one of them runs, no further diagnostic can be issued, which is invariant I4. + +`skip_until_found` will stop at an unmatched closing bracket without consuming it, so a missing `)` does not eat the rest of the function. `skip_to_end_of_block_or_statement` will not: it consumes the semicolon or brace it stops at, so at least one token of your program is discarded even when the parser had already understood it. + +### 3.10 The conflict marker + +```text +function peek_conflict_marker (parser: Parser, first: TokenType) -> Location or nothing + complexity: O(1) + + if peek(parser, 2).type != first + return nothing + if peek(parser, 3).type != first + return nothing + if peek(parser, 4).type != final_kind (first) + return nothing + start = peek(parser, 1).location + if column (start) != 1 + return nothing + return make_location (start, start, finish (peek(parser, 4).location)) +``` + +`c_parser_peek_conflict_marker` at `gcc/c/c-parser.cc:1028@releases/gcc-16.2.0`. `final_kind` maps `CPP_LSHIFT` to `CPP_LESS`, `CPP_RSHIFT` to `CPP_GREATER` and `CPP_EQ_EQ` to `CPP_EQ`. + +Seven identical characters at the start of a line lex as three two character tokens and one single, which is four tokens, which is the width of the buffer in section 2.4. This is the only caller in the C parser that reaches slot four, and it is not parsing C when it does. The column test is what stops `a << b << c << d` from being reported as a merge conflict. + +## 4. Invariants + +**I1.** `parser.tokens_avail` is at most 4 while parsing C, and `peek(parser, n)` is called only with `n <= parser.tokens_avail + 1`. +Established by: the callers. Checked by: `gcc_assert` in `c_parser_peek_nth_token` and `c_parser_peek_2nd_token` at `gcc/c/c-parser.cc:560@releases/gcc-16.2.0`, unconditionally, not only under `--enable-checking`. May be broken by: replay from a pre-lexed token vector, where `tokens_avail` may exceed 4 and `tokens` does not point into `tokens_buf`, for the duration of an OpenMP attribute syntax pragma. + +**I2.** `parser.tokens[0].type` is never `CPP_EOF` when `consume` is called. +Established by: every caller, which peeks before it consumes. Checked by: `gcc_assert` in `c_parser_consume_token`. May be broken by: nobody. A parser that consumes the end of the file would lex past it, and libcpp returns `CPP_EOF` for ever, so the failure would be a hang rather than a crash if this were not checked. + +**I3.** A token's `id_kind` reflects the symbol table as it was when the token was first lexed, not as it is when the token is used. +Established by: `c_lex_one_token`. Checked by: nothing. May be broken by: any scope closing between the peek and the use, which is a real occurrence and the subject of PR67784. Repaired by: `c_parser_maybe_reclassify_token` at the five call sites in section 3.5, and nowhere else, so a construct that closes a scope and does not call it has the bug. + +**I4.** At most one diagnostic is issued between an error and the next recovery point. +Established by: the latch in `c_parser_error_richloc`. Checked by: nothing. May be broken by: any code that calls `error_at` directly rather than going through the parser's reporting functions, which the semantic analysis in `c-decl.cc` and `c-typeck.cc` does constantly. The latch covers syntax errors and nothing else. + +**I5.** A diagnostic that carries a fix-it hint for a missing token has its primary location at the hint and the token that was actually seen as a secondary location. +Established by: the swap in `maybe_suggest_missing_token_insertion`. Checked by: nothing in the compiler, and by `tests/test_cparse.py::test_the_swapped_diagnostic_keeps_the_place_the_caret_came_from` in this project. May be broken by: nobody. This is the observable form of I5 and section 5.2 states how to see it. + +**I6.** `parser.last_token_location` is the location of the most recently consumed token, or `UNKNOWN_LOCATION` before the first. +Established by: `c_parser_consume_token`. Checked by: nothing. May be broken by: `c_parser_consume_pragma` at `gcc/c/c-parser.cc:989@releases/gcc-16.2.0`, which shuffles the window without updating it, so a fix-it hint offered immediately after a pragma is placed relative to the token before the pragma. + +## 5. Observable behaviour + +Everything in this section is recorded in `corpora/diag/f03.json` and read by `gxray.cparse`. Fifteen programs, twenty two diagnostics. Three of the fifteen were also compiled by an x86-64 Linux GCC of the same release through Compiler Explorer, and their diagnostics agree character for character, which is the evidence for the header line saying this component is not target dependent. + +### 5.1 One mistake, thirteen possible sentences + +Eight programs in the recording differ only in the token after a missing semicolon, and produce eight different messages. Corpus entries `brace`, `name`, `number`, `string`, `char`, `keyword`, `pragma`, `eof`. + +| Entry | What follows | Message | +|---|---|---| +| `brace` | `}` | `expected ';' before '}' token` | +| `name` | `b;` | `expected ',' or ';' before 'b'` | +| `number` | `2;` | `expected ';' before numeric constant` | +| `string` | `"s";` | `expected ';' before string constant` | +| `char` | `'c';` | `expected ';' before 'c'` | +| `keyword` | `while` | `expected ';' before 'while'` | +| `pragma` | `#pragma` | `expected ';' before '#pragma'` | +| `eof` | end of file | `expected ';' at end of input` | + +`char` and `keyword` are the same branch of section 3.8 in different disguises: a keyword is reported as `CPP_NAME`, so `'while'` comes from `%qE`. `char` and `name` are two different branches with the same printed form, which is the collision section 3.8 records, and no consumer of the text can separate them. + +### 5.2 Where the caret goes + +Seven of those eight put the caret at column 23, which is the position the semicolon should have occupied, and carry a fix-it hint inserting `;`. One, `name`, puts it at column 25, which is the token that upset the parser, and carries no hint. The two sets coincide exactly: hinted, moved and column 23 are the same seven entries. + +A swapped diagnostic can be recognised without reading GCC's source. Its fix-it hint and its primary location are the same place, and it carries a secondary location somewhere else. `gxray.cparse.Diagnostic.moved` is that test. + +### 5.3 Recovery is visible in the error count + +Corpus entry `recovery` is a function body with three missing semicolons and produces two errors, of which the second is `expected declaration or statement at end of input` pointing at a closing brace that is not wrong. The first mistake is reported at line 4, the recovery in section 3.9 consumes to the end of the block, and the third mistake is never reached. + +### 5.4 Matching brackets are folded into one diagnostic + +Corpus entry `paren` is an unclosed `(`, and its first error carries three locations: a caret at the position the `)` should have been, a secondary under the `(` on the line above, and a secondary under the token that upset the parser on the line below. That is `add_location_if_nearby` succeeding. When it fails, because the two are too far apart to draw together, the same information arrives as a separate `to match this '('` note instead. + +### 5.5 The conflict marker + +Corpus entry `conflict` is a file with git conflict markers in it, and produces three errors reading `version control conflict marker in file`, each with a span seven columns wide starting at column 1. The width is the whole marker, built by `make_location` in section 3.10. + +### 5.6 What a diagnostic carries beyond its text + +`-fdiagnostics-format=sarif-stderr` prints the structure rather than the prose, and is how the recording was made. GCC 16 accepts `text`, `sarif-file` and `sarif-stderr`, and the JSON format that earlier releases accepted is gone. SARIF message strings use braces for placeholders, so GCC doubles any brace in a message, and `expected ';' before '}' token` arrives with `'}}'` in it. `-fdiagnostics-parseable-fixits` prints the same hints in a one line form intended for an editor. + +## 6. Edge cases and error paths + +**Empty translation unit.** A file with no tokens at all is a pedwarn, `ISO C forbids an empty translation unit`, at `gcc/c/c-parser.cc:2085@releases/gcc-16.2.0`, and then the parser returns. It is not an error and the compilation succeeds. + +**End of input.** `CPP_EOF` is returned for ever once reached. The parser never consumes it, by I2, and never peeks past it, because `c_parser_peek_2nd_token` asserts the first token is not `CPP_EOF`. An error reported at end of input gets its caret from `input_location`, which is not updated for `CPP_EOF`, so the message points at the last real token. + +**Peeking out of order.** `peek(parser, 3)` without a preceding `peek(parser, 2)` is `gcc_assert (parser->tokens_avail == n - 1)` failing, which is an internal compiler error with the usual `Please submit a full bug report` text. The assertion is not conditional on `--enable-checking`. + +**A token that cannot be classified.** There is no such case. `c_lex_one_token` gives every `CPP_NAME` an `id_kind`, defaulting to `C_ID_ID`, and every other type gets `C_ID_NONE`. + +**An unknown identifier where a type belongs.** Section 3.4 guesses, and then `c_parser_declaration_or_fndef` recovers by rewriting the token in place, at `gcc/c/c-parser.cc:2555@releases/gcc-16.2.0`: the token's type becomes `CPP_KEYWORD`, its keyword becomes `RID_VOID` and its value becomes `error_mark_node`. Parsing then continues as though you had written `void`, which gets the pointer types right for the rest of the declaration and produces one error instead of a cascade. Nested function definitions are refused after this recovery, on the grounds that a nested function is not what the user is likely to have meant. + +**A pragma in the middle of an expression.** A `CPP_PRAGMA` token can appear anywhere libcpp put one, including places no grammar rule allows. `c_parser_error` reports it through the `%<#pragma%>` branch of section 3.8, and recovery consumes to the `CPP_PRAGMA_EOL`. `parser.in_pragma` stops the recovery routines from running past the end of the pragma into the rest of the statement. + +**Conflict markers.** Recognised before the ordinary error path, in `c_parser_error_richloc`, so the message is about the marker rather than about `<<`. Recognition requires all four lookahead slots and a start column of 1, and a marker indented by one space is reported as a shift expression instead. + +**A second error before recovery.** Discarded silently by the latch. This is not a rate limit and there is no counter: it is one bit, and the next syntax error after it is set produces nothing at all. + +**Errors from semantic analysis.** Not latched. `'x' undeclared` comes from `c-decl.cc` and appears whether or not the parser is recovering, which is why corpus entry `scope-variable` reports an undeclared identifier on a line that the parser read without complaint. + +**Allocation failure.** Not handled. The parser allocates on `parser_obstack` and through the garbage collector, and both abort the compiler rather than returning. + +**Recursion depth.** Unbounded. Nested parentheses recurse in `c_parser_postfix_expression`, and a sufficiently deeply nested expression exhausts the stack. GCC installs a stack overflow handler that turns this into a diagnostic on hosts that support it, which is out of scope for this document. + +## 7. Interactions + +**Reads from libcpp.** Through `c_lex_with_flags` in `gcc/c/c-lex.cc`, which is the front end's wrapper over `cpp_get_token`, and which is responsible for string literal concatenation, execution character set translation and turning preprocessing numbers into constants. `BP-CPP` specifies what arrives. + +**Reads and writes the symbol table.** `lookup_name` in `gcc/c/c-decl.cc` is called from the lexer, from the type name predicate and from the reclassifier, and its answer is what makes C parseable. Everything the parser builds goes back into the same file through `start_decl`, `finish_decl`, `push_scope`, `pop_scope` and their neighbours. + +**Reads the keyword table.** `C_RID_CODE` over the `IDENTIFIER_NODE`, populated by `c_common_init_ts` from `gcc/c-family/c-common.cc`. The parser has no keyword list of its own. + +**Calls one target hook.** `targetm.addr_space.diagnose_usage`, during token classification, for a named address space qualifier. It is the only target dependence in the file. + +**Shares diagnostics with C++.** `c_parse_error` and `maybe_suggest_missing_token_insertion` are both in `gcc/c-family/c-common.cc` and both are called by the C++ parser too, so a change to either changes both languages' messages. + +**Globals it touches.** `the_parser`, the one parser. `input_location`, written by `c_parser_set_source_position_from_token` and read by every diagnostic that does not carry its own location. `parser_obstack`, for scratch. `current_function_decl` and the scope stack, both owned by `c-decl.cc`. `errorcount`, incremented by the diagnostic machinery, and read by the caller of `c_parse_file` to decide whether to proceed. The pseudocode in section 3 passes `parser` explicitly and hides `input_location`; both are globals in the source. + +**Ordering.** Nothing runs before it except libcpp initialisation and precompiled header loading. Everything the middle end does runs after, on the GENERIC it produced, which is `BP-GENERIC` and then `BP-GIMPLE`. + +## 8. Conformance + +**Invariants as assertions.** I1 and I2 are `gcc_assert` calls in the tree and need no test. I3 has no assertion and is tested behaviourally by the PR67784 cases below. I4 and I5 are tested in this project. + +**DejaGnu tests.** + +| Test | What it pins | +|---|---| +| `gcc/testsuite/gcc.dg/pr67784-1.c` | five ways a scope can close between the peek and the use, all reclassified correctly | +| `gcc/testsuite/gcc.dg/pr67784-2.c` | the same five with the typedef and the variable swapped, all producing `undeclared` | +| `gcc/testsuite/gcc.dg/pr67784-3.c` | the same, with the declaration in an `if` rather than a `for` | +| `gcc/testsuite/gcc.dg/pr67784-4.c` | the same, in a `switch` | +| `gcc/testsuite/gcc.dg/pr67784-5.c` | the same, in a `while` | +| `gcc/testsuite/c-c++-common/conflict-markers-1.c` | all three marker kinds recognised, and the lines between them skipped | +| `gcc/testsuite/c-c++-common/conflict-markers-2.c` through `-11.c` | markers in ten more positions, including ones that must not be recognised | +| `gcc/testsuite/gcc.dg/semicolon-fixits.c` | the fix-it hints for extra and missing semicolons, with their multiline caret output | +| `gcc/testsuite/gcc.dg/parse-error-1.c` through `-3.c` | recovery does not cascade | +| `gcc/testsuite/gcc.dg/parse-decl-after-if.c` | a declaration where the reclassifier has to have run | +| `gcc/testsuite/gcc.dg/parse-decl-after-label.c` | the label case of section 3.4 | +| `gcc/testsuite/gcc.dg/parser-pr28152.c` | keywords reported through the identifier branch of section 3.8 | +| `gcc/testsuite/gcc.dg/fixits.c` | parseable fix-it output | + +The conflict marker tests are in `c-c++-common` rather than in `gcc.dg`, which is the testsuite's way of recording that `c_parser_peek_conflict_marker` has a C++ twin with the same behaviour. + +**Golden corpus entries.** `corpora/diag/f03.json`, all fifteen. The recorder asserts thirty conditions about them before writing the file, and `tests/test_cparse.py` asserts fifty six about the file afterwards, of which four compare the tables in section 3.7 and section 3.8 with the pinned tree in both directions. + +**Registered in Tier 0** as `f03-cparse`, kind `offline`. Offline because the comparators count basic blocks and phi nodes in a tree dump, and none of these programs reaches a tree dump. The online half is the three programs compiled on a second target, cached under `tools/cecache/store`. + +## 9. Port notes + +**Recursive descent is not forced.** C is not LL(k) for any k, and the four slot buffer is not what makes GCC's parser work; the symbol table lookup during lexing is. Any parsing technique that can consult a symbol table mid parse will do, and several C compilers use generated LALR parsers with the same hack bolted on. What is forced is that something must decide whether an identifier names a type before the construct containing it can be parsed, because `A * b;` and `A (b);` and `(A) * b` are each two different parses and the token stream does not distinguish them. + +**Four slots is arbitrary.** Three would parse C. The fourth exists for `c_parser_peek_conflict_marker` and nothing else, and a reimplementation that did not want to recognise merge conflicts would need three. A reimplementation that lexed the whole file up front, which is what GCC's own C++ front end does and what the comment at `gcc/c/c-parser.h:33@releases/gcc-16.2.0` wishes the C front end did, would need none. + +**Classifying at lex time is a choice with a cost.** Writing the symbol table's answer into the token means the answer can go stale, which is I3, and GCC pays for that with `c_parser_maybe_reclassify_token` and five call sites that have to be kept in step with the language. Classifying at use time instead costs a symbol table walk per use rather than per token, and removes the bug class. GCC's choice is historical. + +**The error latch is a choice.** One bit means the second syntax error in a statement is never seen. Reporting every error and letting the user filter is also defensible and is what some compilers do. What is not defensible is reporting some of them, which is what happens if the flag is cleared in the wrong place, and the eighteen clear sites in `c-parser.cc` are the surface where that can go wrong. + +**Caret placement is a choice, and the swap is the interesting half.** Pointing at the token that upset the parser is the obvious behaviour and is what GCC does when it has no repair to suggest. Pointing at where the repair would go is better and is what GCC does when it has one, and the cost is that the two cases look inconsistent to anybody who has not read section 3.7. A reimplementation could do either uniformly. Doing the swap requires keeping the location of the previously consumed token, which is I6 and is one field. + +**The unknown type name guess is a choice.** Treating an undeclared identifier followed by another identifier as a misspelled type is a heuristic with no basis in the standard, and it exists so that `foo bar;` gets a useful message. It can be wrong. A reimplementation that omitted it would produce a syntax error instead, which is correct and less helpful. + +**Nothing here is target dependent** except `C_ID_ADDRSPACE` and the one hook in section 7. Two targets compiling the same wrong program produce the same diagnostics, which section 5 records as an observation rather than an assertion. diff --git a/citations.lock.json b/citations.lock.json index 70e0f84..e4e6dee 100644 --- a/citations.lock.json +++ b/citations.lock.json @@ -223,6 +223,34 @@ "hash": "ffc15cd485455b35", "text": "#define EDGE_COMPLEX \\" }, + "gcc/c-family/c-common.cc:10012@releases/gcc-16.2.0": { + "hash": "f208a870e7fc41ef", + "text": " of the diagnostic than that of the following token, so we swap" + }, + "gcc/c-family/c-common.cc:10049@releases/gcc-16.2.0": { + "hash": "2306baafc03ad630", + "text": "maybe_suggest_missing_token_insertion (rich_location *richloc," + }, + "gcc/c-family/c-common.cc:7000@releases/gcc-16.2.0": { + "hash": "95264dfd15df5d5b", + "text": "/* Issue the error given by GMSGID at RICHLOC, indicating that it occurred" + }, + "gcc/c-family/c-common.cc:7004@releases/gcc-16.2.0": { + "hash": "5e83df36043f0f4a", + "text": "c_parse_error (const char *gmsgid, enum cpp_ttype token_type," + }, + "gcc/c-family/c-common.cc:9973@releases/gcc-16.2.0": { + "hash": "5981bdafce48626d", + "text": "static enum missing_token_insertion_kind" + }, + "gcc/c-family/c-common.cc:9974@releases/gcc-16.2.0": { + "hash": "44bfca4a488da10a", + "text": "get_missing_token_insertion_kind (enum cpp_ttype type)" + }, + "gcc/c-family/c-common.cc:9999@releases/gcc-16.2.0": { + "hash": "176afbb8416f084c", + "text": "/* Given RICHLOC, a location for a diagnostic describing a missing token" + }, "gcc/c-family/c-opts.cc:1419@releases/gcc-16.2.0": { "hash": "4de52375ef4441f9", "text": "c_common_parse_file (void)" @@ -287,10 +315,186 @@ "hash": "7c20d8a7b3e8dcc2", "text": "cc1$(exeext): $(C_OBJS) cc1-checksum.o $(BACKEND) $(LIBDEPS)" }, + "gcc/c/c-parser.cc:1006@releases/gcc-16.2.0": { + "hash": "c7f898a1ddae03e5", + "text": "/* Update the global input_location from TOKEN. */" + }, + "gcc/c/c-parser.cc:1009@releases/gcc-16.2.0": { + "hash": "3872780dafdd4cda", + "text": "{" + }, + "gcc/c/c-parser.cc:1016@releases/gcc-16.2.0": { + "hash": "bb9fa74ab690e7a6", + "text": "/* Helper function for c_parser_error." + }, + "gcc/c/c-parser.cc:1028@releases/gcc-16.2.0": { + "hash": "57174bb777ddc489", + "text": "c_parser_peek_conflict_marker (c_parser *parser, enum cpp_ttype tok1_kind," + }, + "gcc/c/c-parser.cc:1072@releases/gcc-16.2.0": { + "hash": "8a7019d1df6a382e", + "text": "static bool" + }, + "gcc/c/c-parser.cc:1073@releases/gcc-16.2.0": { + "hash": "649e19e26ce9a16f", + "text": "c_parser_error_richloc (c_parser *parser, const char *gmsgid," + }, + "gcc/c/c-parser.cc:1122@releases/gcc-16.2.0": { + "hash": "e79de6dbd86c7486", + "text": "\t\t /* ??? The C parser does not save the cpp flags of a" + }, + "gcc/c/c-parser.cc:1130@releases/gcc-16.2.0": { + "hash": "d0cef0903d1dcfeb", + "text": "/* As c_parser_error_richloc, but issue the message at the" + }, + "gcc/c/c-parser.cc:1135@releases/gcc-16.2.0": { + "hash": "105ef13dc342f00a", + "text": "c_parser_error (c_parser *parser, const char *gmsgid)" + }, + "gcc/c/c-parser.cc:1278@releases/gcc-16.2.0": { + "hash": "08949a8b3bced6c4", + "text": "bool" + }, + "gcc/c/c-parser.cc:1279@releases/gcc-16.2.0": { + "hash": "8de3d8cfe9457e28", + "text": "c_parser_require (c_parser *parser," + }, + "gcc/c/c-parser.cc:1353@releases/gcc-16.2.0": { + "hash": "390b536b7e8ed625", + "text": "c_parser_skip_until_found (c_parser *parser," + }, + "gcc/c/c-parser.cc:1426@releases/gcc-16.2.0": { + "hash": "c3143c6b99e2181c", + "text": "c_parser_skip_to_end_of_parameter (c_parser *parser)" + }, + "gcc/c/c-parser.cc:1507@releases/gcc-16.2.0": { + "hash": "34b90798af0934e2", + "text": "c_parser_skip_to_pragma_eol (c_parser *parser, bool error_if_not_eol = true)" + }, + "gcc/c/c-parser.cc:1594@releases/gcc-16.2.0": { + "hash": "0bd58e4454e69b5d", + "text": "c_parser_skip_to_end_of_block_or_statement (c_parser *parser," + }, + "gcc/c/c-parser.cc:188@releases/gcc-16.2.0": { + "hash": "9f79021941720d97", + "text": "/* A parser structure recording information about the state and" + }, + "gcc/c/c-parser.cc:191@releases/gcc-16.2.0": { + "hash": "463cd632449d62c3", + "text": "struct GTY(()) c_parser {" + }, + "gcc/c/c-parser.cc:2065@releases/gcc-16.2.0": { + "hash": "b8ece7b5e725b924", + "text": "/* Parse a translation unit (C90 6.7, C99 6.9, C11 6.9)." + }, + "gcc/c/c-parser.cc:2081@releases/gcc-16.2.0": { + "hash": "7e4c1547c7397703", + "text": "c_parser_translation_unit (c_parser *parser)" + }, + "gcc/c/c-parser.cc:2085@releases/gcc-16.2.0": { + "hash": "1fa718b246183ee5", + "text": " pedwarn (c_parser_peek_token (parser)->location, OPT_Wpedantic," + }, + "gcc/c/c-parser.cc:2321@releases/gcc-16.2.0": { + "hash": "e66ecdba240d3864", + "text": "/* We might need to reclassify any previously-lexed identifier, e.g." + }, + "gcc/c/c-parser.cc:2326@releases/gcc-16.2.0": { + "hash": "70df294a168aa298", + "text": "c_parser_maybe_reclassify_token (c_parser *parser)" + }, + "gcc/c/c-parser.cc:2555@releases/gcc-16.2.0": { + "hash": "4191d582b7ebe4b3", + "text": " c_parser_peek_token (parser)->keyword = RID_VOID;" + }, "gcc/c/c-parser.cc:31269@releases/gcc-16.2.0": { "hash": "7c397c5b5864c371", "text": "c_parse_file (void)" }, + "gcc/c/c-parser.cc:31308@releases/gcc-16.2.0": { + "hash": "c13571568c037c31", + "text": "/* Parse the body of a function declaration marked with \"__RTL\"." + }, + "gcc/c/c-parser.cc:332@releases/gcc-16.2.0": { + "hash": "6976a3871192bc94", + "text": "/* Read in and lex a single token, storing it in *TOKEN. If RAW," + }, + "gcc/c/c-parser.cc:336@releases/gcc-16.2.0": { + "hash": "fce39af86a7fea56", + "text": "c_lex_one_token (c_parser *parser, c_token *token, bool raw = false)" + }, + "gcc/c/c-parser.cc:392@releases/gcc-16.2.0": { + "hash": "0295a26e7cd64a49", + "text": "\t\ttargetm.addr_space.diagnose_usage (as, token->location);" + }, + "gcc/c/c-parser.cc:397@releases/gcc-16.2.0": { + "hash": "bea14c5370f2ce81", + "text": "\t else if (c_dialect_objc () && OBJC_IS_PQ_KEYWORD (rid_code))" + }, + "gcc/c/c-parser.cc:467@releases/gcc-16.2.0": { + "hash": "a0cd5ee4fd18837d", + "text": "\tdecl = lookup_name (token->value);" + }, + "gcc/c/c-parser.cc:560@releases/gcc-16.2.0": { + "hash": "c9c42da458ce83f7", + "text": " gcc_assert (parser->tokens_avail == 1);" + }, + "gcc/c/c-parser.cc:572@releases/gcc-16.2.0": { + "hash": "f17eb553b149f233", + "text": "c_parser_peek_nth_token (c_parser *parser, unsigned int n)" + }, + "gcc/c/c-parser.cc:696@releases/gcc-16.2.0": { + "hash": "01458ee8e46155b2", + "text": "c_parser_next_tokens_start_typename (c_parser *parser, enum c_lookahead_kind la," + }, + "gcc/c/c-parser.cc:8907@releases/gcc-16.2.0": { + "hash": "983a720982acd2df", + "text": " c_parser_maybe_reclassify_token (parser);" + }, + "gcc/c/c-parser.cc:916@releases/gcc-16.2.0": { + "hash": "8497fcbe318ed33b", + "text": "c_parser_next_tokens_start_declaration (c_parser *parser, unsigned int n)" + }, + "gcc/c/c-parser.cc:958@releases/gcc-16.2.0": { + "hash": "8c2ee53f5ecc60f9", + "text": "c_parser_consume_token (c_parser *parser)" + }, + "gcc/c/c-parser.cc:969@releases/gcc-16.2.0": { + "hash": "6476c4952e2fdf99", + "text": " if (parser->tokens != &parser->tokens_buf[0])" + }, + "gcc/c/c-parser.cc:989@releases/gcc-16.2.0": { + "hash": "625ae4af6974b072", + "text": "c_parser_consume_pragma (c_parser *parser)" + }, + "gcc/c/c-parser.h:118@releases/gcc-16.2.0": { + "hash": "cade0a885acb4ab4", + "text": "enum c_parser_prec {" + }, + "gcc/c/c-parser.h:133@releases/gcc-16.2.0": { + "hash": "5114ba3835cff388", + "text": "enum c_lookahead_kind {" + }, + "gcc/c/c-parser.h:26@releases/gcc-16.2.0": { + "hash": "9afab8fbdf0fc5c7", + "text": "/* The C lexer intermediates between the lexer in cpplib and c-lex.cc" + }, + "gcc/c/c-parser.h:33@releases/gcc-16.2.0": { + "hash": "0e2e0e35a66726af", + "text": " ??? It might be a good idea to lex the whole file up front (as for" + }, + "gcc/c/c-parser.h:38@releases/gcc-16.2.0": { + "hash": "84e9c103902800bf", + "text": "enum c_id_kind {" + }, + "gcc/c/c-parser.h:4@releases/gcc-16.2.0": { + "hash": "bd461c820a7dbf39", + "text": " Parser actions based on the old Bison parser; structure somewhat" + }, + "gcc/c/c-parser.h:53@releases/gcc-16.2.0": { + "hash": "336965a723c96c23", + "text": "struct GTY (()) c_token {" + }, "gcc/cfg.cc:169@releases/gcc-16.2.0": { "hash": "2be7cc40efdca37e", "text": "compact_blocks (void)" @@ -2251,6 +2455,10 @@ "hash": "fd113b9d7b63acea", "text": "struct GTY(()) cpp_macro {" }, + "libcpp/include/rich-location.h:620@releases/gcc-16.2.0": { + "hash": "df359a55f2ced7b9", + "text": "class fixit_hint" + }, "libcpp/init.cc:235@releases/gcc-16.2.0": { "hash": "0ce93728bb8bc698", "text": " CPP_OPTION (pfile, max_include_depth) = 200;" diff --git a/corpora/diag/f03.json b/corpora/diag/f03.json new file mode 100644 index 0000000..a5f4af3 --- /dev/null +++ b/corpora/diag/f03.json @@ -0,0 +1,856 @@ +{ + "recorded": "2026-09-06", + "tag": "releases/gcc-16.2.0", + "compiler": "gcc-16 (Homebrew GCC 16.2.0) 16.2.0", + "target": "aarch64-apple-darwin24", + "cache": [ + "2abc7c5ac5fcd1dfa3a01d67b8464971", + "54c4cc6a52013b9b4c21da0c7e74a9bd", + "b671a775fa867d6a251b39a9a885057d", + "feded2677aff4cb36cb66975aa68b30e" + ], + "cases": { + "brace": { + "about": "the next token is a close brace", + "flags": [], + "source": "int f(void) { return 1 }\n", + "text": "p.c: In function 'f':\np.c:1:23: error: expected ';' before '}' token\n 1 | int f(void) { return 1 }\n | ^~\n | ;\n", + "diagnostics": [ + { + "level": "error", + "message": "expected ';' before '}' token", + "at": [ + 1, + 23, + 24 + ], + "snippet": "int f(void) { return 1 }", + "function": "f", + "fixes": [ + { + "at": [ + 1, + 23, + 23 + ], + "insert": ";" + } + ], + "related": [ + [ + 1, + 24, + 25 + ] + ] + } + ], + "elsewhere": "p.c: In function 'f':\np.c:1:23: error: expected ';' before '}' token\n 1 | int f(void) { return 1 }\n | ^~\n | ;" + }, + "name": { + "about": "the next token is an identifier", + "flags": [], + "source": "int f(void) { int a = 1 b; }\n", + "text": "p.c: In function 'f':\np.c:1:25: error: expected ',' or ';' before 'b'\n 1 | int f(void) { int a = 1 b; }\n | ^\n", + "diagnostics": [ + { + "level": "error", + "message": "expected ',' or ';' before 'b'", + "at": [ + 1, + 25, + 26 + ], + "snippet": "int f(void) { int a = 1 b; }", + "function": "f" + } + ] + }, + "number": { + "about": "the next token is a number", + "flags": [], + "source": "int f(void) { return 1 2; }\n", + "text": "p.c: In function 'f':\np.c:1:23: error: expected ';' before numeric constant\n 1 | int f(void) { return 1 2; }\n | ^~\n | ;\n", + "diagnostics": [ + { + "level": "error", + "message": "expected ';' before numeric constant", + "at": [ + 1, + 23, + 24 + ], + "snippet": "int f(void) { return 1 2; }", + "function": "f", + "fixes": [ + { + "at": [ + 1, + 23, + 23 + ], + "insert": ";" + } + ], + "related": [ + [ + 1, + 24, + 25 + ] + ] + } + ] + }, + "string": { + "about": "the next token is a string", + "flags": [], + "source": "int f(void) { return 1 \"x\"; }\n", + "text": "p.c: In function 'f':\np.c:1:23: error: expected ';' before string constant\n 1 | int f(void) { return 1 \"x\"; }\n | ^~~~\n | ;\n", + "diagnostics": [ + { + "level": "error", + "message": "expected ';' before string constant", + "at": [ + 1, + 23, + 24 + ], + "snippet": "int f(void) { return 1 \"x\"; }", + "function": "f", + "fixes": [ + { + "at": [ + 1, + 23, + 23 + ], + "insert": ";" + } + ], + "related": [ + [ + 1, + 24, + 27 + ] + ] + } + ] + }, + "char": { + "about": "the next token is a character constant", + "flags": [], + "source": "int f(void) { return 1 'c'; }\n", + "text": "p.c: In function 'f':\np.c:1:23: error: expected ';' before 'c'\n 1 | int f(void) { return 1 'c'; }\n | ^~~~\n | ;\n", + "diagnostics": [ + { + "level": "error", + "message": "expected ';' before 'c'", + "at": [ + 1, + 23, + 24 + ], + "snippet": "int f(void) { return 1 'c'; }", + "function": "f", + "fixes": [ + { + "at": [ + 1, + 23, + 23 + ], + "insert": ";" + } + ], + "related": [ + [ + 1, + 24, + 27 + ] + ] + } + ] + }, + "keyword": { + "about": "the next token is a keyword", + "flags": [], + "source": "int f(void) { return 1 while (0); }\n", + "text": "p.c: In function 'f':\np.c:1:23: error: expected ';' before 'while'\n 1 | int f(void) { return 1 while (0); }\n | ^~~~~~\n | ;\n", + "diagnostics": [ + { + "level": "error", + "message": "expected ';' before 'while'", + "at": [ + 1, + 23, + 24 + ], + "snippet": "int f(void) { return 1 while (0); }", + "function": "f", + "fixes": [ + { + "at": [ + 1, + 23, + 23 + ], + "insert": ";" + } + ], + "related": [ + [ + 1, + 24, + 29 + ] + ] + } + ] + }, + "pragma": { + "about": "the next token is a pragma, which is one token", + "flags": [], + "source": "int f(void) { return 1\n#pragma GCC unroll 2\n; }\n", + "text": "p.c: In function 'f':\np.c:1:23: error: expected ';' before '#pragma'\n 1 | int f(void) { return 1\n | ^\n | ;\n 2 | #pragma GCC unroll 2\n | ~~~ \n", + "diagnostics": [ + { + "level": "error", + "message": "expected ';' before '#pragma'", + "at": [ + 1, + 23, + 24 + ], + "snippet": "int f(void) { return 1", + "function": "f", + "fixes": [ + { + "at": [ + 1, + 23, + 23 + ], + "insert": ";" + } + ], + "related": [ + [ + 2, + 9, + 12 + ] + ] + } + ] + }, + "eof": { + "about": "there is no next token", + "flags": [], + "source": "int f(void) { return 1\n", + "text": "p.c: In function 'f':\np.c:1:23: error: expected ';' at end of input\n 1 | int f(void) { return 1\n | ^\n | ;\np.c:1:1: error: expected declaration or statement at end of input\n 1 | int f(void) { return 1\n | ^~~\n", + "diagnostics": [ + { + "level": "error", + "message": "expected ';' at end of input", + "at": [ + 1, + 23, + 24 + ], + "snippet": "int f(void) { return 1", + "function": "f", + "fixes": [ + { + "at": [ + 1, + 23, + 23 + ], + "insert": ";" + } + ], + "related": [ + [ + 2, + 1, + 0 + ] + ] + }, + { + "level": "error", + "message": "expected declaration or statement at end of input", + "at": [ + 1, + 1, + 4 + ], + "snippet": "int f(void) { return 1", + "function": "f" + } + ] + }, + "meaning-typedef": { + "about": "A is a type, so line 3 declares b", + "flags": [ + "-Wall", + "-Wshadow" + ], + "source": "typedef int A;\nint b;\nvoid f(void) { A * b; }\n", + "text": "p.c: In function 'f':\np.c:3:20: warning: declaration of 'b' shadows a global declaration [-Wshadow]\n 3 | void f(void) { A * b; }\n | ^\np.c:2:5: note: shadowed declaration is here\n 2 | int b;\n | ^\np.c:3:20: warning: unused variable 'b' [-Wunused-variable]\n 3 | void f(void) { A * b; }\n | ^\n", + "diagnostics": [ + { + "level": "warning", + "message": "declaration of 'b' shadows a global declaration", + "at": [ + 3, + 20, + 21 + ], + "snippet": "void f(void) { A * b; }", + "function": "f", + "related": [ + [ + 2, + 5, + 6 + ] + ] + }, + { + "level": "warning", + "message": "unused variable 'b'", + "at": [ + 3, + 20, + 21 + ], + "snippet": "void f(void) { A * b; }", + "function": "f" + } + ], + "elsewhere": "p.c: In function 'f':\np.c:3:20: warning: declaration of 'b' shadows a global declaration [-Wshadow]\n 3 | void f(void) { A * b; }\n | ^\np.c:2:5: note: shadowed declaration is here\n 2 | int b;\n | ^\np.c:3:20: warning: unused variable 'b' [-Wunused-variable]\n 3 | void f(void) { A * b; }\n | ^" + }, + "meaning-variable": { + "about": "A is a variable, so the same line 3 multiplies", + "flags": [ + "-Wall", + "-Wshadow" + ], + "source": "int A;\nint b;\nvoid f(void) { A * b; }\n", + "text": "p.c: In function 'f':\np.c:3:18: warning: statement with no effect [-Wunused-value]\n 3 | void f(void) { A * b; }\n | ~~^~~\n", + "diagnostics": [ + { + "level": "warning", + "message": "statement with no effect", + "at": [ + 3, + 16, + 21 + ], + "snippet": "void f(void) { A * b; }", + "function": "f" + } + ], + "elsewhere": "p.c: In function 'f':\np.c:3:18: warning: statement with no effect [-Wunused-value]\n 3 | void f(void) { A * b; }\n | ~~^~~" + }, + "scope-typedef": { + "about": "T is a type at file scope and a variable inside a for header that has closed", + "flags": [ + "-Wall" + ], + "source": "typedef int T;\nvoid f(void)\n{\n for (int T;;)\n if (1)\n ;\n T *x;\n}\n", + "text": "p.c: In function 'f':\np.c:4:12: warning: unused variable 'T' [-Wunused-variable]\n 4 | for (int T;;)\n | ^\np.c:7:6: warning: unused variable 'x' [-Wunused-variable]\n 7 | T *x;\n | ^\n", + "diagnostics": [ + { + "level": "warning", + "message": "unused variable 'T'", + "at": [ + 4, + 12, + 13 + ], + "snippet": " for (int T;;)", + "function": "f" + }, + { + "level": "warning", + "message": "unused variable 'x'", + "at": [ + 7, + 6, + 7 + ], + "snippet": " T *x;", + "function": "f" + } + ] + }, + "scope-variable": { + "about": "The same program with the two declarations of T swapped over", + "flags": [ + "-Wall" + ], + "source": "int T;\nvoid f(void)\n{\n for (typedef int T;;)\n if (1)\n ;\n T *x;\n}\n", + "text": "p.c: In function 'f':\np.c:7:6: error: 'x' undeclared (first use in this function)\n 7 | T *x;\n | ^\np.c:7:6: note: each undeclared identifier is reported only once for each function it appears in\n", + "diagnostics": [ + { + "level": "error", + "message": "'x' undeclared (first use in this function)", + "at": [ + 7, + 6, + 7 + ], + "snippet": " T *x;", + "function": "f", + "related": [ + [ + 7, + 6, + 7 + ] + ] + } + ] + }, + "recovery": { + "about": "Three missing semicolons, and not three errors", + "flags": [], + "source": "void f(void)\n{\n int a = 1\n int b = 2\n int c = 3\n}\n", + "text": "p.c: In function 'f':\np.c:4:3: error: expected ',' or ';' before 'int'\n 4 | int b = 2\n | ^~~\np.c:6:1: error: expected declaration or statement at end of input\n 6 | }\n | ^\n", + "diagnostics": [ + { + "level": "error", + "message": "expected ',' or ';' before 'int'", + "at": [ + 4, + 3, + 6 + ], + "snippet": " int b = 2", + "function": "f" + }, + { + "level": "error", + "message": "expected declaration or statement at end of input", + "at": [ + 6, + 1, + 2 + ], + "snippet": "}", + "function": "f" + } + ] + }, + "paren": { + "about": "A bracket that never closes, and the second place the message points", + "flags": [], + "source": "void g(void);\nvoid f(int x)\n{\n if (x\n g();\n}\n", + "text": "p.c: In function 'f':\np.c:4:8: error: expected ')' before 'g'\n 4 | if (x\n | ~ ^\n | )\n 5 | g();\n | ~ \np.c:6:1: error: expected expression before '}' token\n 6 | }\n | ^\n", + "diagnostics": [ + { + "level": "error", + "message": "expected ')' before 'g'", + "at": [ + 4, + 8, + 9 + ], + "snippet": " if (x", + "function": "f", + "fixes": [ + { + "at": [ + 4, + 8, + 8 + ], + "insert": ")" + } + ], + "related": [ + [ + 5, + 5, + 6 + ], + [ + 4, + 6, + 7 + ] + ] + }, + { + "level": "error", + "message": "expected expression before '}' token", + "at": [ + 6, + 1, + 2 + ], + "snippet": "}", + "function": "f" + } + ] + }, + "conflict": { + "about": "The deepest the C parser ever looks ahead, and what it looks for", + "flags": [], + "source": "int f(void)\n{\n<<<<<<< HEAD\n return 1;\n=======\n return 2;\n>>>>>>> other\n}\n", + "text": "p.c: In function 'f':\np.c:3:1: error: version control conflict marker in file\n 3 | <<<<<<< HEAD\n | ^~~~~~~\np.c:5:1: error: version control conflict marker in file\n 5 | =======\n | ^~~~~~~\np.c:7:1: error: version control conflict marker in file\n 7 | >>>>>>> other\n | ^~~~~~~\n", + "diagnostics": [ + { + "level": "error", + "message": "version control conflict marker in file", + "at": [ + 3, + 1, + 8 + ], + "snippet": "<<<<<<< HEAD", + "function": "f" + }, + { + "level": "error", + "message": "version control conflict marker in file", + "at": [ + 5, + 1, + 8 + ], + "snippet": "=======", + "function": "f" + }, + { + "level": "error", + "message": "version control conflict marker in file", + "at": [ + 7, + 1, + 8 + ], + "snippet": ">>>>>>> other", + "function": "f" + } + ] + } + }, + "grammar": { + "functions": [ + "c_parser_alignas_specifier", + "c_parser_alignof_expression", + "c_parser_all_labels", + "c_parser_asm_clobbers", + "c_parser_asm_definition", + "c_parser_asm_goto_operands", + "c_parser_asm_operands", + "c_parser_asm_statement", + "c_parser_asm_string_literal", + "c_parser_attribute_arguments", + "c_parser_balanced_token_sequence", + "c_parser_bc_name", + "c_parser_binary_expression", + "c_parser_braced_init", + "c_parser_c99_block_statement", + "c_parser_cast_expression", + "c_parser_check_balanced_raw_token_sequence", + "c_parser_check_literal_zero", + "c_parser_compound_literal_scspecs", + "c_parser_compound_statement", + "c_parser_compound_statement_nostart", + "c_parser_condition", + "c_parser_conditional_expression", + "c_parser_consume_pragma", + "c_parser_consume_token", + "c_parser_declaration_or_fndef", + "c_parser_declarator", + "c_parser_declspecs", + "c_parser_direct_declarator", + "c_parser_direct_declarator_inner", + "c_parser_do_statement", + "c_parser_else_body", + "c_parser_enum_specifier", + "c_parser_error", + "c_parser_error_richloc", + "c_parser_expr_list", + "c_parser_expr_no_commas", + "c_parser_expression", + "c_parser_expression_conv", + "c_parser_external_declaration", + "c_parser_for_statement", + "c_parser_generic_selection", + "c_parser_get_builtin_args", + "c_parser_gnu_attribute", + "c_parser_gnu_attribute_any_word", + "c_parser_gnu_attributes", + "c_parser_handle_directive_omp_attributes", + "c_parser_handle_musttail", + "c_parser_handle_statement_omp_attributes", + "c_parser_has_attribute_expression", + "c_parser_if_body", + "c_parser_if_statement", + "c_parser_initelt", + "c_parser_initializer", + "c_parser_initval", + "c_parser_label", + "c_parser_maxof_or_minof_expression", + "c_parser_maybe_reclassify_token", + "c_parser_next_token_is_qualifier", + "c_parser_next_token_starts_declspecs", + "c_parser_next_tokens_start_declaration", + "c_parser_next_tokens_start_typename", + "c_parser_nth_token_starts_std_attributes", + "c_parser_oacc_all_clauses", + "c_parser_oacc_cache", + "c_parser_oacc_clause_async", + "c_parser_oacc_clause_tile", + "c_parser_oacc_clause_wait", + "c_parser_oacc_compute", + "c_parser_oacc_compute_clause_self", + "c_parser_oacc_data", + "c_parser_oacc_data_clause", + "c_parser_oacc_data_clause_deviceptr", + "c_parser_oacc_declare", + "c_parser_oacc_enter_exit_data", + "c_parser_oacc_host_data", + "c_parser_oacc_loop", + "c_parser_oacc_routine", + "c_parser_oacc_shape_clause", + "c_parser_oacc_simple_clause", + "c_parser_oacc_single_int_clause", + "c_parser_oacc_update", + "c_parser_oacc_wait", + "c_parser_oacc_wait_list", + "c_parser_objc_alias_declaration", + "c_parser_objc_at_dynamic_declaration", + "c_parser_objc_at_property_declaration", + "c_parser_objc_at_synthesize_declaration", + "c_parser_objc_class_declaration", + "c_parser_objc_class_definition", + "c_parser_objc_class_instance_variables", + "c_parser_objc_diagnose_bad_element_prefix", + "c_parser_objc_keywordexpr", + "c_parser_objc_maybe_method_attributes", + "c_parser_objc_message_args", + "c_parser_objc_method_decl", + "c_parser_objc_method_definition", + "c_parser_objc_method_type", + "c_parser_objc_methodproto", + "c_parser_objc_methodprotolist", + "c_parser_objc_protocol_definition", + "c_parser_objc_protocol_refs", + "c_parser_objc_receiver", + "c_parser_objc_selector", + "c_parser_objc_selector_arg", + "c_parser_objc_synchronized_statement", + "c_parser_objc_try_catch_finally_statement", + "c_parser_objc_type_name", + "c_parser_omp_all_clauses", + "c_parser_omp_allocate", + "c_parser_omp_assume", + "c_parser_omp_assumes", + "c_parser_omp_assumption_clauses", + "c_parser_omp_atomic", + "c_parser_omp_barrier", + "c_parser_omp_begin", + "c_parser_omp_cancel", + "c_parser_omp_cancellation_point", + "c_parser_omp_clause_affinity", + "c_parser_omp_clause_aligned", + "c_parser_omp_clause_allocate", + "c_parser_omp_clause_bind", + "c_parser_omp_clause_branch", + "c_parser_omp_clause_cancelkind", + "c_parser_omp_clause_collapse", + "c_parser_omp_clause_copyin", + "c_parser_omp_clause_copyprivate", + "c_parser_omp_clause_default", + "c_parser_omp_clause_defaultmap", + "c_parser_omp_clause_depend", + "c_parser_omp_clause_destroy", + "c_parser_omp_clause_detach", + "c_parser_omp_clause_device", + "c_parser_omp_clause_device_type", + "c_parser_omp_clause_dist_schedule", + "c_parser_omp_clause_doacross", + "c_parser_omp_clause_doacross_sink", + "c_parser_omp_clause_dyn_groupprivate", + "c_parser_omp_clause_filter", + "c_parser_omp_clause_final", + "c_parser_omp_clause_firstprivate", + "c_parser_omp_clause_from_to", + "c_parser_omp_clause_full", + "c_parser_omp_clause_grainsize", + "c_parser_omp_clause_has_device_addr", + "c_parser_omp_clause_hint", + "c_parser_omp_clause_if", + "c_parser_omp_clause_indirect", + "c_parser_omp_clause_init", + "c_parser_omp_clause_init_modifiers", + "c_parser_omp_clause_interop", + "c_parser_omp_clause_is_device_ptr", + "c_parser_omp_clause_lastprivate", + "c_parser_omp_clause_linear", + "c_parser_omp_clause_map", + "c_parser_omp_clause_mergeable", + "c_parser_omp_clause_name", + "c_parser_omp_clause_nocontext", + "c_parser_omp_clause_nogroup", + "c_parser_omp_clause_nontemporal", + "c_parser_omp_clause_novariants", + "c_parser_omp_clause_nowait", + "c_parser_omp_clause_num_tasks", + "c_parser_omp_clause_num_teams", + "c_parser_omp_clause_num_threads", + "c_parser_omp_clause_order", + "c_parser_omp_clause_ordered", + "c_parser_omp_clause_orderedkind", + "c_parser_omp_clause_partial", + "c_parser_omp_clause_priority", + "c_parser_omp_clause_private", + "c_parser_omp_clause_proc_bind", + "c_parser_omp_clause_reduction", + "c_parser_omp_clause_safelen", + "c_parser_omp_clause_schedule", + "c_parser_omp_clause_shared", + "c_parser_omp_clause_simdlen", + "c_parser_omp_clause_thread_limit", + "c_parser_omp_clause_uniform", + "c_parser_omp_clause_untied", + "c_parser_omp_clause_use", + "c_parser_omp_clause_use_device_addr", + "c_parser_omp_clause_use_device_ptr", + "c_parser_omp_clause_uses_allocators", + "c_parser_omp_construct", + "c_parser_omp_context_selector", + "c_parser_omp_context_selector_specification", + "c_parser_omp_critical", + "c_parser_omp_declare", + "c_parser_omp_declare_mapper", + "c_parser_omp_declare_reduction", + "c_parser_omp_declare_simd", + "c_parser_omp_declare_target", + "c_parser_omp_depobj", + "c_parser_omp_directive_args", + "c_parser_omp_dispatch", + "c_parser_omp_dispatch_body", + "c_parser_omp_distribute", + "c_parser_omp_end", + "c_parser_omp_error", + "c_parser_omp_flush", + "c_parser_omp_for", + "c_parser_omp_for_loop", + "c_parser_omp_groupprivate", + "c_parser_omp_interop", + "c_parser_omp_iterators", + "c_parser_omp_loop", + "c_parser_omp_loop_nest", + "c_parser_omp_masked", + "c_parser_omp_master", + "c_parser_omp_metadirective", + "c_parser_omp_modifier_prefer_type", + "c_parser_omp_next_tokens_can_be_canon_loop", + "c_parser_omp_nothing", + "c_parser_omp_ordered", + "c_parser_omp_parallel", + "c_parser_omp_requires", + "c_parser_omp_scan_loop_body", + "c_parser_omp_scope", + "c_parser_omp_section_scan", + "c_parser_omp_sections", + "c_parser_omp_sections_scope", + "c_parser_omp_sequence_args", + "c_parser_omp_simd", + "c_parser_omp_single", + "c_parser_omp_structured_block", + "c_parser_omp_structured_block_sequence", + "c_parser_omp_target", + "c_parser_omp_target_data", + "c_parser_omp_target_enter_data", + "c_parser_omp_target_exit_data", + "c_parser_omp_target_update", + "c_parser_omp_task", + "c_parser_omp_taskgroup", + "c_parser_omp_taskloop", + "c_parser_omp_taskwait", + "c_parser_omp_taskyield", + "c_parser_omp_teams", + "c_parser_omp_threadprivate", + "c_parser_omp_tile", + "c_parser_omp_tile_sizes", + "c_parser_omp_unroll", + "c_parser_omp_var_list_parens", + "c_parser_omp_variable_list", + "c_parser_parameter_declaration", + "c_parser_paren_condition", + "c_parser_paren_selection_header", + "c_parser_parms_declarator", + "c_parser_parms_list_declarator", + "c_parser_parse_rtl_body", + "c_parser_peek_2nd_token", + "c_parser_peek_conflict_marker", + "c_parser_peek_nth_token", + "c_parser_peek_nth_token_raw", + "c_parser_peek_token", + "c_parser_postfix_expression", + "c_parser_postfix_expression_after_paren_type", + "c_parser_postfix_expression_after_primary", + "c_parser_pragma", + "c_parser_pragma_pch_preprocess", + "c_parser_pragma_unroll", + "c_parser_predefined_identifier", + "c_parser_require", + "c_parser_require_keyword", + "c_parser_selection_header", + "c_parser_set_error", + "c_parser_set_source_position_from_token", + "c_parser_simple_asm_expr", + "c_parser_sizeof_or_countof_expression", + "c_parser_skip_std_attribute_spec_seq", + "c_parser_skip_to_closing_brace", + "c_parser_skip_to_end_of_block_or_statement", + "c_parser_skip_to_end_of_parameter", + "c_parser_skip_to_pragma_eol", + "c_parser_skip_to_pragma_omp_end_declare_variant", + "c_parser_skip_until_found", + "c_parser_statement", + "c_parser_statement_after_labels", + "c_parser_static_assert_declaration", + "c_parser_static_assert_declaration_no_semi", + "c_parser_std_attribute", + "c_parser_std_attribute_list", + "c_parser_std_attribute_specifier", + "c_parser_std_attribute_specifier_sequence", + "c_parser_string_literal", + "c_parser_struct_declaration", + "c_parser_struct_or_union_specifier", + "c_parser_switch_statement", + "c_parser_tokens_buf", + "c_parser_transaction", + "c_parser_transaction_attributes", + "c_parser_transaction_cancel", + "c_parser_transaction_expression", + "c_parser_translation_unit", + "c_parser_type_name", + "c_parser_typeof_specifier", + "c_parser_unary_expression", + "c_parser_while_statement" + ] + }, + "lookahead": { + "peeks": 729, + "seconds": 99, + "depths": { + "2": 2, + "3": 6, + "4": 3 + }, + "slots": 4 + } +} diff --git a/corpora/source/f03.json b/corpora/source/f03.json new file mode 100644 index 0000000..3d18d3d --- /dev/null +++ b/corpora/source/f03.json @@ -0,0 +1,448 @@ +{ + "tag": "releases/gcc-16.2.0", + "snippets": { + "intermediates": { + "path": "gcc/c/c-parser.h", + "first": 26, + "last": 35, + "about": "What sits between libcpp and the parser, and the wish at the end of it", + "lines": [ + "/* The C lexer intermediates between the lexer in cpplib and c-lex.cc", + " and the C parser. Unlike the C++ lexer, the parser structure", + " stores the lexer information instead of using a separate structure.", + " Identifiers are separated into ordinary identifiers, type names,", + " keywords and some other Objective-C types of identifiers, and some", + " look-ahead is maintained.", + "", + " ??? It might be a good idea to lex the whole file up front (as for", + " C++). It would then be possible to share more of the C and C++", + " lexer code, if desired. */" + ] + }, + "slots": { + "path": "gcc/c/c-parser.cc", + "first": 188, + "last": 198, + "about": "The parser's entire memory of your file, and the comment that undercounts it", + "lines": [ + "/* A parser structure recording information about the state and", + " context of parsing. Includes lexer information with up to two", + " tokens of look-ahead; more are not needed for C. */", + "struct GTY(()) c_parser {", + " /* The look-ahead tokens. */", + " c_token * GTY((skip)) tokens;", + " /* Buffer for look-ahead tokens. */", + " c_token tokens_buf[4];", + " /* How many look-ahead tokens are available (0 - 4, or", + " more if parsing from pre-lexed tokens). */", + " unsigned int tokens_avail;" + ] + }, + "handover": { + "path": "gcc/c/c-parser.cc", + "first": 332, + "last": 349, + "about": "The one call. Everything the parser will ever know arrives through it", + "lines": [ + "/* Read in and lex a single token, storing it in *TOKEN. If RAW,", + " context-sensitive postprocessing of the token is not done. */", + "", + "static void", + "c_lex_one_token (c_parser *parser, c_token *token, bool raw = false)", + "{", + " timevar_push (TV_LEX);", + "", + " if (raw || vec_safe_length (parser->raw_tokens) == 0)", + " {", + " token->type = c_lex_with_flags (&token->value, &token->location,", + "\t\t\t\t &token->flags,", + "\t\t\t\t (parser->lex_joined_string", + "\t\t\t\t ? 0 : C_LEX_STRING_NO_JOIN));", + " token->id_kind = C_ID_NONE;", + " token->keyword = RID_MAX;", + " token->pragma_kind = PRAGMA_NONE;", + " }" + ] + }, + "lookup": { + "path": "gcc/c/c-parser.cc", + "first": 467, + "last": 491, + "about": "The symbol table decides what an identifier is, while it is being lexed", + "lines": [ + "\tdecl = lookup_name (token->value);", + "\tif (decl)", + "\t {", + "\t if (TREE_CODE (decl) == TYPE_DECL)", + "\t {", + "\t\ttoken->id_kind = C_ID_TYPENAME;", + "\t\tbreak;", + "\t }", + "\t }", + "\telse if (c_dialect_objc ())", + "\t {", + "\t tree objc_interface_decl = objc_is_class_name (token->value);", + "\t /* Objective-C class names are in the same namespace as", + "\t variables and typedefs, and hence are shadowed by local", + "\t declarations. */", + "\t if (objc_interface_decl", + " && (!objc_force_identifier || global_bindings_p ()))", + "\t {", + "\t\ttoken->value = objc_interface_decl;", + "\t\ttoken->id_kind = C_ID_CLASSNAME;", + "\t\tbreak;", + "\t }", + "\t }", + " token->id_kind = C_ID_ID;", + " }" + ] + }, + "conflict": { + "path": "gcc/c/c-parser.cc", + "first": 1016, + "last": 1054, + "about": "The only thing in C that needs a fourth token of lookahead", + "lines": [ + "/* Helper function for c_parser_error.", + " Having peeked a token of kind TOK1_KIND that might signify", + " a conflict marker, peek successor tokens to determine", + " if we actually do have a conflict marker.", + " Specifically, we consider a run of 7 '<', '=' or '>' characters", + " at the start of a line as a conflict marker.", + " These come through the lexer as three pairs and a single,", + " e.g. three CPP_LSHIFT (\"<<\") and a CPP_LESS ('<').", + " If it returns true, *OUT_LOC is written to with the location/range", + " of the marker. */", + "", + "static bool", + "c_parser_peek_conflict_marker (c_parser *parser, enum cpp_ttype tok1_kind,", + "\t\t\t location_t *out_loc)", + "{", + " c_token *token2 = c_parser_peek_2nd_token (parser);", + " if (token2->type != tok1_kind)", + " return false;", + " c_token *token3 = c_parser_peek_nth_token (parser, 3);", + " if (token3->type != tok1_kind)", + " return false;", + " c_token *token4 = c_parser_peek_nth_token (parser, 4);", + " if (token4->type != conflict_marker_get_final_tok_kind (tok1_kind))", + " return false;", + "", + " /* It must be at the start of the line. */", + " location_t start_loc = c_parser_peek_token (parser)->location;", + " if (LOCATION_COLUMN (start_loc) != 1)", + " return false;", + "", + " /* We have a conflict marker. Construct a location of the form:", + " <<<<<<<", + " ^~~~~~~", + " with start == caret, finishing at the end of the marker. */", + " location_t finish_loc = get_finish (token4->location);", + " *out_loc = make_location (start_loc, start_loc, finish_loc);", + "", + " return true;", + "}" + ] + }, + "position": { + "path": "gcc/c/c-parser.cc", + "first": 1006, + "last": 1014, + "about": "The guard that decides where an at-end-of-input message puts its caret", + "lines": [ + "/* Update the global input_location from TOKEN. */", + "static inline void", + "c_parser_set_source_position_from_token (c_token *token)", + "{", + " if (token->type != CPP_EOF)", + " {", + " input_location = token->location;", + " }", + "}" + ] + }, + "latch": { + "path": "gcc/c/c-parser.cc", + "first": 1072, + "last": 1094, + "about": "Two lines, and the reason three mistakes are not three errors", + "lines": [ + "static bool", + "c_parser_error_richloc (c_parser *parser, const char *gmsgid,", + "\t\t\trich_location *richloc)", + "{", + " c_token *token = c_parser_peek_token (parser);", + " if (parser->error)", + " return false;", + " parser->error = true;", + " if (!gmsgid)", + " return false;", + "", + " /* If this is actually a conflict marker, report it as such. */", + " if (token->type == CPP_LSHIFT", + " || token->type == CPP_RSHIFT", + " || token->type == CPP_EQ_EQ)", + " {", + " location_t loc;", + " if (c_parser_peek_conflict_marker (parser, token->type, &loc))", + "\t{", + "\t error_at (loc, \"version control conflict marker in file\");", + "\t return true;", + "\t}", + " }" + ] + }, + "report": { + "path": "gcc/c/c-parser.cc", + "first": 1130, + "last": 1141, + "about": "The other way to report, which puts the caret on the token it can see", + "lines": [ + "/* As c_parser_error_richloc, but issue the message at the", + " location of PARSER's next token, or at input_location", + " if the next token is EOF. */", + "", + "bool", + "c_parser_error (c_parser *parser, const char *gmsgid)", + "{", + " c_token *token = c_parser_peek_token (parser);", + " c_parser_set_source_position_from_token (token);", + " rich_location richloc (line_table, input_location);", + " return c_parser_error_richloc (parser, gmsgid, &richloc);", + "}" + ] + }, + "require": { + "path": "gcc/c/c-parser.cc", + "first": 1278, + "last": 1318, + "about": "Where the caret goes when GCC knows which token you left out", + "lines": [ + "bool", + "c_parser_require (c_parser *parser,", + "\t\t enum cpp_ttype type,", + "\t\t const char *msgid,", + "\t\t location_t matching_location,", + "\t\t bool type_is_unique)", + "{", + " if (c_parser_next_token_is (parser, type))", + " {", + " c_parser_consume_token (parser);", + " return true;", + " }", + " else", + " {", + " location_t next_token_loc = c_parser_peek_token (parser)->location;", + " gcc_rich_location richloc (next_token_loc);", + "", + " /* Potentially supply a fix-it hint, suggesting to add the", + "\t missing token immediately after the *previous* token.", + "\t This may move the primary location within richloc. */", + " if (!parser->error && type_is_unique)", + "\tmaybe_suggest_missing_token_insertion (&richloc, type,", + "\t\t\t\t\t parser->last_token_location);", + "", + " /* If matching_location != UNKNOWN_LOCATION, highlight it.", + "\t Attempt to consolidate diagnostics by printing it as a", + "\t secondary range within the main diagnostic. */", + " bool added_matching_location = false;", + " if (matching_location != UNKNOWN_LOCATION)", + "\tadded_matching_location", + "\t = richloc.add_location_if_nearby (*global_dc, matching_location);", + "", + " if (c_parser_error_richloc (parser, msgid, &richloc))", + "\t/* If we weren't able to consolidate matching_location, then", + "\t print it as a secondary diagnostic. */", + "\tif (matching_location != UNKNOWN_LOCATION && !added_matching_location)", + "\t inform (matching_location, \"to match this %qs\",", + "\t\t get_matching_symbol (type));", + "", + " return false;", + " }" + ] + }, + "loop": { + "path": "gcc/c/c-parser.cc", + "first": 2065, + "last": 2099, + "about": "The whole of the C parser's top level, which is six lines", + "lines": [ + "/* Parse a translation unit (C90 6.7, C99 6.9, C11 6.9).", + "", + " translation-unit:", + " external-declarations", + "", + " external-declarations:", + " external-declaration", + " external-declarations external-declaration", + "", + " GNU extensions:", + "", + " translation-unit:", + " empty", + "*/", + "", + "static void", + "c_parser_translation_unit (c_parser *parser)", + "{", + " if (c_parser_next_token_is (parser, CPP_EOF))", + " {", + " pedwarn (c_parser_peek_token (parser)->location, OPT_Wpedantic,", + "\t \"ISO C forbids an empty translation unit\");", + " }", + " else", + " {", + " void *obstack_position = obstack_alloc (&parser_obstack, 0);", + " mark_valid_location_for_stdc_pragma (false);", + " do", + "\t{", + "\t ggc_collect ();", + "\t c_parser_external_declaration (parser);", + "\t obstack_free (&parser_obstack, obstack_position);", + "\t}", + " while (c_parser_next_token_is_not (parser, CPP_EOF));", + " }" + ] + }, + "reclassify": { + "path": "gcc/c/c-parser.cc", + "first": 2321, + "last": 2341, + "about": "The patch-up a bug report bought, and the scope it puts right", + "lines": [ + "/* We might need to reclassify any previously-lexed identifier, e.g.", + " when we've left a for loop with an if-statement without else in the", + " body - we might have used a wrong scope for the token. See PR67784. */", + "", + "static void", + "c_parser_maybe_reclassify_token (c_parser *parser)", + "{", + " if (c_parser_next_token_is (parser, CPP_NAME))", + " {", + " c_token *token = c_parser_peek_token (parser);", + "", + " if (token->id_kind == C_ID_ID || token->id_kind == C_ID_TYPENAME)", + "\t{", + "\t tree decl = lookup_name (token->value);", + "", + "\t token->id_kind = C_ID_ID;", + "\t if (decl)", + "\t {", + "\t if (TREE_CODE (decl) == TYPE_DECL)", + "\t\ttoken->id_kind = C_ID_TYPENAME;", + "\t }" + ] + }, + "suffixes": { + "path": "gcc/c-family/c-common.cc", + "first": 7000, + "last": 7013, + "about": "Where a parser's complaint becomes a sentence, and the first branch of it", + "lines": [ + "/* Issue the error given by GMSGID at RICHLOC, indicating that it occurred", + " before TOKEN, which had the associated VALUE. */", + "", + "void", + "c_parse_error (const char *gmsgid, enum cpp_ttype token_type,", + "\t tree value, unsigned char token_flags,", + "\t rich_location *richloc)", + "{", + "#define catenate_messages(M1, M2) catenate_strings ((M1), (M2), sizeof (M2))", + "", + " char *message = NULL;", + "", + " if (token_type == CPP_EOF)", + " message = catenate_messages (gmsgid, \" at end of input\");" + ] + }, + "insertion": { + "path": "gcc/c-family/c-common.cc", + "first": 9973, + "last": 9997, + "about": "The seven tokens GCC will offer to write for you, and which side they go", + "lines": [ + "static enum missing_token_insertion_kind", + "get_missing_token_insertion_kind (enum cpp_ttype type)", + "{", + " switch (type)", + " {", + " /* Insert missing \"opening\" brackets immediately", + "\t before the next token. */", + " case CPP_OPEN_SQUARE:", + " case CPP_OPEN_PAREN:", + " return MTIK_INSERT_BEFORE_NEXT;", + "", + " /* Insert other missing symbols immediately after", + "\t the previous token. */", + " case CPP_CLOSE_PAREN:", + " case CPP_CLOSE_SQUARE:", + " case CPP_SEMICOLON:", + " case CPP_COMMA:", + " case CPP_COLON:", + " return MTIK_INSERT_AFTER_PREV;", + "", + " /* Other kinds of token don't get fix-it hints. */", + " default:", + " return MTIK_IMPOSSIBLE;", + " }", + "}" + ] + }, + "swap": { + "path": "gcc/c-family/c-common.cc", + "first": 9999, + "last": 10046, + "about": "GCC explaining, in a comment with a diagram, why the caret is on the line above", + "lines": [ + "/* Given RICHLOC, a location for a diagnostic describing a missing token", + " of kind TOKEN_TYPE, potentially add a fix-it hint suggesting the", + " insertion of the token.", + "", + " The location of the attempted fix-it hint depends on TOKEN_TYPE:", + " it will either be:", + " (a) immediately after PREV_TOKEN_LOC, or", + "", + " (b) immediately before the primary location within RICHLOC (taken to", + "\t be that of the token following where the token was expected).", + "", + " If we manage to add a fix-it hint, then the location of the", + " fix-it hint is likely to be more useful as the primary location", + " of the diagnostic than that of the following token, so we swap", + " these locations.", + "", + " For example, given this bogus code:", + " 123456789012345678901234567890", + " 1 | int missing_semicolon (void)", + " 2 | {", + " 3 | return 42", + " 4 | }", + "", + " we will emit:", + "", + " \"expected ';' before '}'\"", + "", + " RICHLOC's primary location is at the closing brace, so before \"swapping\"", + " we would emit the error at line 4 column 1:", + "", + " 123456789012345678901234567890", + " 3 | return 42 |< fix-it hint emitted for this line", + " | ; |", + " 4 | } |< \"expected ';' before '}'\" emitted at this line", + " | ^ |", + "", + " It's more useful for the location of the diagnostic to be at the", + " fix-it hint, so we swap the locations, so the primary location", + " is at the fix-it hint, with the old primary location inserted", + " as a secondary location, giving this, with the error at line 3", + " column 12:", + "", + " 123456789012345678901234567890", + " 3 | return 42 |< \"expected ';' before '}'\" emitted at this line,", + " | ^ | with fix-it hint", + " 4 | ; |", + " | } |< secondary range emitted here", + " | ~ |. */" + ] + } + } +} diff --git a/docs/blueprints.md b/docs/blueprints.md index 8601afa..5daeafa 100644 --- a/docs/blueprints.md +++ b/docs/blueprints.md @@ -13,6 +13,7 @@ The pseudocode every algorithm is written in is [NOTATION](blueprints/NOTATION.m | [BP-BOOTSTRAP](blueprints/BP-BOOTSTRAP.md) | building the compiler with itself | partial | 2 of 9 | no | | [BP-BUILD](blueprints/BP-BUILD.md) | configuring and building the compiler | partial | 2 of 9 | yes | | [BP-CFG](blueprints/BP-CFG.md) | blocks and edges | stub | none | no | +| [BP-CPARSE](blueprints/BP-CPARSE.md) | the C parser | partial | none | no | | [BP-CPP](blueprints/BP-CPP.md) | the preprocessor | partial | none | yes | | [BP-DEBUGGING](blueprints/BP-DEBUGGING.md) | stopping the compiler and looking at it | complete | 2 of 9 | no | | [BP-DRIVER](blueprints/BP-DRIVER.md) | the program that runs the other programs | partial | none | yes | @@ -26,7 +27,7 @@ The pseudocode every algorithm is written in is [NOTATION](blueprints/NOTATION.m | [BP-SSA](blueprints/BP-SSA.md) | one definition per name | stub | none | no | | [BP-TESTSUITE](blueprints/BP-TESTSUITE.md) | running GCC's own tests | partial | 2 of 9 | yes | -15 of 58 written: 2 complete, 7 partial, 6 stub. +16 of 58 written: 2 complete, 8 partial, 6 stub. ## What each one is for @@ -42,6 +43,10 @@ This document specifies what happens between a source tree and a working `cc1`. This is a stub. It holds the data structures, the two fixed blocks, how a GIMPLE sequence becomes a graph, and what dominance is computed by and when it is valid. The pass level detail is not here: loop discovery, profile propagation, hot and cold partitioning, and the RTL side of the hooks are named and not specified. Section 2.3 could be generated from `gcc/cfg-flags.def`, which is a `.def` file with exactly the shape `bpc` reads, and that is the obvious first thing to do when this stub is promoted. +### [BP-CPARSE](blueprints/BP-CPARSE.md), the C parser + +This document specifies the C front end's parser: the code in `gcc/c/c-parser.cc` and `gcc/c/c-parser.h` that turns the token stream libcpp produced into calls on the tree building interface in `gcc/c/c-decl.cc` and `gcc/c/c-typeck.cc`. The token, the parser state, the four slot lookahead buffer and the identifier classification are specified field by field. Lexing one token, peeking, consuming, the three disambiguations C cannot make without a symbol table, error reporting, fix-it insertion, caret placement and the four recovery routines are specified as algorithms. Semantic analysis, the tree building interface, attributes, OpenMP, OpenACC, Objective-C, transactional memory and the `__RTL` and `__GIMPLE` function body parsers are named and not specified, and each place that stops short says so. Nothing here is generated, because the parser keeps no tables in `.def` files: its grammar is control flow and its keyword set is a C enum. What exists instead is `gxray.cparse`, a reader for recorded diagnostics, with tests that compare its transcription of `get_missing_token_insertion_kind` and of the thirteen `c_parse_error` branches against the pinned tree, so a GCC that grows a case fails the build rather than making a paragraph quietly false. + ### [BP-CPP](blueprints/BP-CPP.md), the preprocessor This document specifies libcpp, the library that turns a source file into a stream of preprocessing tokens, and the client in `gcc/c-family/c-ppoutput.cc` that prints that stream when you ask for `-E`. The token, the token type table, the hash node and the macro are specified field by field. Lexing, macro expansion, argument prescan, stringification, pasting, the disabling rule, the multiple include optimization and the spacing rule the printer applies are specified as algorithms. Traditional mode, modules and header units, precompiled headers, `#embed`, character set conversion and the `#if` expression evaluator are named and not specified, and each place that stops short says so. Nothing here is generated, because libcpp keeps its tables in C macros rather than in `.def` files a script could read without a C parser. What exists instead is `gxray.cpp`, a reader for recorded preprocessor output, with tests that compare its table of paste-avoiding pairs against the case labels of `cpp_avoid_paste` in the pinned tree, so a GCC that grows a pair fails the build rather than making a paragraph quietly false. diff --git a/docs/blueprints/BP-CPARSE.md b/docs/blueprints/BP-CPARSE.md new file mode 100644 index 0000000..3af6783 --- /dev/null +++ b/docs/blueprints/BP-CPARSE.md @@ -0,0 +1,3 @@ + + +--8<-- "blueprints/BP-CPARSE.md" diff --git a/docs/lessons.md b/docs/lessons.md index d68377b..ee634e4 100644 --- a/docs/lessons.md +++ b/docs/lessons.md @@ -25,5 +25,6 @@ Every one is a notebook with a Colab badge, so there is nothing to install. Part | B05 | [Sixty lines of C++, and you are inside the compiler](https://github.com/tamnd/gcc-internals/blob/main/lessons/b05-the-plugin/b05.ipynb) | GCC's plugin mechanism from the outside in: the three things a plugin has to have, the three ways it is refused, what an event actually is, a GIMPLE pass of your own inserted after ssa with its own dump file, switching one of GCC's passes off from outside and watching the assembly move, and why the thing you are writing against is not an API | M2 | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/tamnd/gcc-internals/blob/main/lessons/b05-the-plugin/b05.ipynb) | | F01 | [The driver is an interpreter](https://github.com/tamnd/gcc-internals/blob/main/lessons/f01-the-spec-language/f01.ipynb) | That `gcc -dumpspecs` prints a program, in a language with conditionals and function calls, which the driver interprets to decide what to run; how to read it; and four text files that change what your compiler does without patching or rebuilding it | M3 | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/tamnd/gcc-internals/blob/main/lessons/f01-the-spec-language/f01.ipynb) | | F02 | [The preprocessor is not a text editor](https://github.com/tamnd/gcc-internals/blob/main/lessons/f02-tokens-not-text/f02.ipynb) | That the preprocessor lexes your file into tokens before it does anything else, and that its printed output is a rendering of those tokens rather than the tokens themselves; the space GCC inserts that is in no input file; the four hundred macros you did not write; and why one #include of stdio.h opens thirty eight files | M3 | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/tamnd/gcc-internals/blob/main/lessons/f02-tokens-not-text/f02.ipynb) | +| F03 | [The C parser can see four tokens](https://github.com/tamnd/gcc-internals/blob/main/lessons/f03-four-tokens/f03.ipynb) | That GCC's C parser is hand written recursive descent whose whole memory is four token slots and the symbol table; why one missing semicolon produces eight different messages; why the caret is on the line above the mistake; why three mistakes come out as two errors; and why `A * b;` needs a symbol table to read | M3 | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/tamnd/gcc-internals/blob/main/lessons/f03-four-tokens/f03.ipynb) | -19 of 96 written. +20 of 96 written. diff --git a/gxray/cparse.py b/gxray/cparse.py new file mode 100644 index 0000000..dded61a --- /dev/null +++ b/gxray/cparse.py @@ -0,0 +1,575 @@ +"""What GCC's C parser saw, read back from the diagnostics it printed. + + from gxray import cparse + + rec = cparse.load("f03") + one = rec.case("brace") + + one.text # the message a person sees, carets and all + one.errors[0].message # "expected ';' before '}' token" + one.errors[0].at # line 3, column 11, which is not where '}' is + one.errors[0].fixes[0].insert # ";" + +A C parser is hard to look at. It has no dump flag, it produces no file, and the tree it +builds is F04's subject rather than this one's. What it does produce, on every program that +is wrong, is a diagnostic, and a GCC diagnostic is a far more detailed readout of parser +state than it looks. It carries the message, the token the parser was looking at, the place +it thinks the mistake is, a suggested repair, and sometimes a second location it wants you +to compare against. Those five things are enough to reconstruct what the parser knew. + +GCC 16 will print them as SARIF, which is a machine readable format meant for static +analysis tools, and this module reads that. Only the `results` array is kept. The rest of a +SARIF log is the absolute path of the compiler, the arguments it was invoked with and two +timestamps, none of which would be the same twice and all of which would make a recording +that cannot be compared with another recording. + +One wrinkle worth knowing about, because it shows up in the recorded text. SARIF message +strings use braces for placeholders, so GCC doubles any brace in a message before writing +it out. `expected ';' before '}' token` arrives as `expected ';' before '}}' token`, and +`undouble` puts it back. The plain text GCC prints alongside is not affected. +""" + +from __future__ import annotations + +import json +import re +from dataclasses import dataclass, field +from pathlib import Path + +CORPUS = Path(__file__).resolve().parent.parent / "corpora" / "diag" + + +class CParseError(Exception): + """Something asked of a recording that the recording cannot answer.""" + + +# --------------------------------------------------------------------------- +# The two tables the C front end decides messages with +# --------------------------------------------------------------------------- + +#: Where a fix-it hint for a missing token goes, transcribed from +#: `get_missing_token_insertion_kind` at gcc/c-family/c-common.cc:9973. Seven token types +#: get a hint at all. An opening bracket goes before the token the parser is looking at, +#: because `if( flag)` is not what anybody meant to write. Everything else goes after the +#: token before it, which is why the caret in `expected ';' before '}' token` lands on the +#: line above the brace. Every other token type gets no hint and no caret move. +INSERTION: dict[str, str] = { + "[": "before", + "(": "before", + ")": "after", + "]": "after", + ";": "after", + ",": "after", + ":": "after", +} + + +@dataclass(frozen=True) +class Suffix: + """One branch of `c_parse_error`, which is the sentence factory for the whole front end. + + A parser function passes in a complaint like `expected ';'` and nothing else. The rest + of the sentence is chosen here, by the type of the token the parser is looking at and by + nothing else. That is why the same mistake reads eight different ways depending on what + happens to come after it. + """ + + #: What GCC appends, with its own format directives left in, exactly as the source has it. + text: str + #: What that turns into once printed, as a pattern anchored to the end of the sentence. + #: Two of these overlap, which is a fact about the messages and not a defect here. + shows: str + #: The token types that select it, spelled as `cpplib.h` spells them. Empty on the last + #: branch, which is a range test rather than a list. + types: tuple[str, ...] + #: What a reader would call the thing, for a table in a notebook. + about: str + + def matches(self, message: str) -> bool: + return re.search(self.shows, message) is not None + + +#: Every branch of `c_parse_error`, in source order, from gcc/c-family/c-common.cc:7003. +#: Thirteen of them, and the last is the one almost every punctuation mark lands in. +SUFFIXES: tuple[Suffix, ...] = ( + Suffix(" at end of input", r" at end of input$", ("CPP_EOF",), "the file ran out"), + Suffix( + " before %s'%c'", + r" before (?:L|u|U|u8)?'.'$", + ("CPP_CHAR", "CPP_WCHAR", "CPP_CHAR16", "CPP_CHAR32", "CPP_UTF8CHAR"), + "a character constant you could print", + ), + Suffix( + " before %s'\\x%x'", + r" before (?:L|u|U|u8)?'\\x[0-9a-f]+'$", + ("CPP_CHAR", "CPP_WCHAR", "CPP_CHAR16", "CPP_CHAR32", "CPP_UTF8CHAR"), + "a character constant you could not", + ), + Suffix( + " before user-defined character literal", + r" before user-defined character literal$", + ( + "CPP_CHAR_USERDEF", + "CPP_WCHAR_USERDEF", + "CPP_CHAR16_USERDEF", + "CPP_CHAR32_USERDEF", + "CPP_UTF8CHAR_USERDEF", + ), + "a C++ thing the C parser can still be handed", + ), + Suffix( + " before user-defined string literal", + r" before user-defined string literal$", + ( + "CPP_STRING_USERDEF", + "CPP_WSTRING_USERDEF", + "CPP_STRING16_USERDEF", + "CPP_STRING32_USERDEF", + "CPP_UTF8STRING_USERDEF", + ), + "the same, in string form", + ), + Suffix( + " before string constant", + r" before string constant$", + ("CPP_STRING", "CPP_WSTRING", "CPP_STRING16", "CPP_STRING32", "CPP_UTF8STRING"), + "a string, whose text is not quoted back at you", + ), + Suffix(" before numeric constant", r" before numeric constant$", ("CPP_NUMBER",), "a number"), + Suffix( + " before %qE", + r" before '[A-Za-z_$][A-Za-z_$0-9]*'$", + ("CPP_NAME",), + "an identifier, which is quoted back at you", + ), + Suffix(" before %<#pragma%>", r" before '#pragma'$", ("CPP_PRAGMA",), "a pragma"), + Suffix( + " before end of line", r" before end of line$", ("CPP_PRAGMA_EOL",), "the end of a pragma" + ), + Suffix( + " before %", r" before 'decltype'$", ("CPP_DECLTYPE",), "a token only C++ makes" + ), + Suffix(" before %<#embed%>", r" before '#embed'$", ("CPP_EMBED",), "a C23 directive"), + Suffix( + " before %qs token", + r" before '.*' token$", + (), + "anything else, which is where punctuation lands", + ), +) + +#: The doubling SARIF does to braces, because `{` starts a placeholder in a SARIF message. +_DOUBLED = re.compile(r"([{}])\1") + +#: A terminal colour escape. Compiler Explorer runs GCC with colour turned on, so a +#: diagnostic that arrives from there has a dozen of these per line and would compare equal +#: to nothing. +_COLOUR = re.compile(r"\x1b\[[0-9;]*[A-Za-z]") + + +def undouble(message: str) -> str: + """Undo SARIF's brace doubling, so a recorded message reads as GCC printed it.""" + return _DOUBLED.sub(r"\1", message) + + +def plain(text: str) -> str: + """A diagnostic with the colour taken out, which is what makes two of them comparable.""" + return _COLOUR.sub("", text) + + +def suffixes_for(message: str) -> list[Suffix]: + """Every branch of `c_parse_error` that could have produced the tail of this message. + + Matching on the printed sentence rather than on a token type, because the printed + sentence is what a recording has, and it is also what a reader has. Usually the answer + is one branch. It is two when the message ends in a single quoted character, because a + one-letter identifier and a character constant print the same and no amount of care + here can separate them. + """ + return [one for one in SUFFIXES if one.matches(message)] + + +def suffix_for(message: str) -> Suffix | None: + """The one branch that produced this message, or nothing if the message does not say.""" + found = suffixes_for(message) + return found[0] if len(found) == 1 else None + + +# --------------------------------------------------------------------------- +# One diagnostic +# --------------------------------------------------------------------------- + + +@dataclass(frozen=True) +class Span: + """A place in a file, the way a diagnostic carries one. + + Columns are one based and `end` is one past the last character, which is SARIF's + convention and also the one the caret line follows: a span three columns wide prints + `^~~`. + """ + + line: int + column: int + end: int = 0 + + @property + def width(self) -> int: + return max(1, self.end - self.column) + + def __str__(self) -> str: + return f"{self.line}:{self.column}" + (f"-{self.end}" if self.end else "") + + +@dataclass(frozen=True) +class Fix: + """A repair GCC is confident enough to write out. + + `insert` is the text, and the span is where it goes. A fix-it with an empty `insert` is + a deletion, which the C parser does not produce for a missing token but other passes do. + """ + + at: Span + insert: str + + def __str__(self) -> str: + return f"insert {self.insert!r} at {self.at}" + + +@dataclass(frozen=True) +class Diagnostic: + """One thing GCC said, with everywhere it pointed while saying it.""" + + level: str + message: str + at: Span + #: The line of your program the caret sits under, as GCC quoted it back. + snippet: str = "" + #: The enclosing function, when GCC knew one. This is the `In function 'f':` line. + function: str = "" + fixes: tuple[Fix, ...] = () + #: Other places the same diagnostic pointed at. For a missing token this is the token + #: the parser was actually looking at, which is usually not where the caret went. + related: tuple[Span, ...] = () + + @property + def error(self) -> bool: + return self.level == "error" + + @property + def suffix(self) -> Suffix | None: + """Which branch of `c_parse_error` chose the tail of this sentence, if it says.""" + return suffix_for(self.message) + + @property + def suffixes(self) -> list[Suffix]: + """Every branch that could have. More than one means the message is not telling.""" + return suffixes_for(self.message) + + @property + def moved(self) -> bool: + """Whether the caret is somewhere other than the token being complained about. + + True when the diagnostic carries a related location on a different line or column + from the caret, which is what the swap in `maybe_suggest_missing_token_insertion` + leaves behind. + """ + return any((one.line, one.column) != (self.at.line, self.at.column) for one in self.related) + + def __str__(self) -> str: + return f"{self.at}: {self.level}: {self.message}" + + +def parse_sarif(text: str) -> list[Diagnostic]: + """Every result in a SARIF log, and nothing else from it. + + Raises if the text is not SARIF, rather than returning nothing, because a recorder that + silently records zero diagnostics for a program that does not compile is a recorder that + has stopped recording. + """ + try: + log = json.loads(text) + except json.JSONDecodeError as exc: + raise CParseError(f"not a SARIF log: {exc}") from exc + runs = log.get("runs") + if not runs: + raise CParseError("a SARIF log with no runs in it") + return [_result(one) for one in runs[0].get("results", [])] + + +def _region(where: dict) -> Span: + physical = where.get("physicalLocation", where) + region = physical.get("region", {}) + return Span( + line=region.get("startLine", 0), + column=region.get("startColumn", 1), + end=region.get("endColumn", 0), + ) + + +def _snippet(where: dict) -> str: + physical = where.get("physicalLocation", where) + context = physical.get("contextRegion", {}) + return context.get("snippet", {}).get("text", "").rstrip("\n") + + +def _result(raw: dict) -> Diagnostic: + locations = raw.get("locations") or [{}] + first = locations[0] + logical = first.get("logicalLocations") or [{}] + fixes = [] + for fix in raw.get("fixes", []): + for change in fix.get("artifactChanges", []): + for one in change.get("replacements", []): + fixes.append( + Fix( + at=_region({"region": one.get("deletedRegion", {})}), + insert=one.get("insertedContent", {}).get("text", ""), + ) + ) + return Diagnostic( + level=raw.get("level", "error"), + message=undouble(raw.get("message", {}).get("text", "")), + at=_region(first), + snippet=_snippet(first), + function=logical[0].get("fullyQualifiedName", ""), + fixes=tuple(fixes), + related=tuple(_region(one) for one in raw.get("relatedLocations", [])), + ) + + +# --------------------------------------------------------------------------- +# One program put through the parser +# --------------------------------------------------------------------------- + + +@dataclass(frozen=True) +class Case: + """A small program, what GCC said about it, and what GCC said in machine form. + + `text` and `diagnostics` are two renderings of one compilation, not two compilations. + The first is what a person reads and the second is what this module can assert against, + and having both is the only way a notebook can show the message and check it. + """ + + name: str + about: str + source: str + text: str + diagnostics: tuple[Diagnostic, ...] = () + #: The same program through an x86-64 Linux GCC of the same release, when the recorder + #: asked for one. A parser is target independent and this is how that gets checked + #: rather than asserted. + elsewhere: str = "" + + @property + def errors(self) -> list[Diagnostic]: + return [one for one in self.diagnostics if one.level == "error"] + + @property + def warnings(self) -> list[Diagnostic]: + return [one for one in self.diagnostics if one.level == "warning"] + + @property + def clean(self) -> bool: + return not self.diagnostics + + @property + def agrees(self) -> bool: + """Whether the other target said the same thing, ignoring trailing space.""" + if not self.elsewhere: + return True + return _flat(self.text) == _flat(self.elsewhere) + + def line(self, number: int) -> str: + """One line of the program, numbered the way a diagnostic numbers it.""" + lines = self.source.split("\n") + if not 1 <= number <= len(lines): + raise CParseError( + f"{self.name} has {len(lines)} lines, and line {number} was asked for" + ) + return lines[number - 1] + + def under(self, one: Diagnostic) -> str: + """The caret line, rebuilt from the span, the way GCC would draw it. + + Rebuilt rather than cut out of `text`, so that a notebook can point at a span GCC + chose not to draw, which is how a related location gets shown next to the caret it + was not printed with. + """ + return " " * (one.at.column - 1) + "^" + "~" * (one.at.width - 1) + + def __str__(self) -> str: + return f"{self.name}: {len(self.errors)} errors, {len(self.warnings)} warnings" + + +def _flat(text: str) -> list[str]: + return [line.rstrip() for line in text.split("\n") if line.strip()] + + +# --------------------------------------------------------------------------- +# What the pinned tree says about itself +# --------------------------------------------------------------------------- + + +@dataclass(frozen=True) +class Grammar: + """The size of the C parser, counted rather than described. + + `functions` is every `c_parser_*` defined in `gcc/c/c-parser.cc`. The rest split that + list by what the function is for, which is the number worth knowing: the file is mostly + not about C. + """ + + functions: tuple[str, ...] = () + + def named(self, *prefixes: str) -> list[str]: + wanted = tuple(f"c_parser_{p}" for p in prefixes) + return sorted(one for one in self.functions if one.startswith(wanted)) + + @property + def dialects(self) -> dict[str, list[str]]: + """The parser functions grouped by which language they are for.""" + groups = { + "OpenMP": self.named("omp"), + "OpenACC": self.named("oacc"), + "Objective-C": self.named("objc"), + "transactional memory": self.named("transaction"), + } + spoken = {name for names in groups.values() for name in names} + return {"C": sorted(set(self.functions) - spoken), **groups} + + def __len__(self) -> int: + return len(self.functions) + + +@dataclass(frozen=True) +class Lookahead: + """Every place the parser looks past the token in front of it. + + `depths` counts calls to `c_parser_peek_nth_token` by the constant they pass, which is + the only way the parser can reach past the second token. `peeks` and `seconds` are the + two cheap helpers, counted for scale rather than for detail. + """ + + peeks: int = 0 + seconds: int = 0 + depths: dict[int, int] = field(default_factory=dict) + #: The buffer in `struct c_parser`, which is the hard limit while parsing C. + slots: int = 0 + + @property + def deepest(self) -> int: + return max(self.depths) if self.depths else 2 + + +@dataclass +class Recording: + """One run of the F03 recorder.""" + + recorded: str + compiler: str + target: str + cases: dict[str, Case] + grammar: Grammar + lookahead: Lookahead + + def case(self, name: str) -> Case: + if name not in self.cases: + have = ", ".join(sorted(self.cases)) + raise CParseError(f"no case called {name!r}. There is: {have}") + return self.cases[name] + + def __getitem__(self, name: str) -> Case: + return self.case(name) + + def __iter__(self): + return iter(self.cases.values()) + + def __len__(self) -> int: + return len(self.cases) + + +def load(name: str = "f03", root: Path | str | None = None) -> Recording: + target = Path(root or CORPUS) / f"{name}.json" + if not target.is_file(): + raise CParseError(f"{target} is not there. Run lessons/f03-four-tokens/record.py.") + raw = json.loads(target.read_text(encoding="utf-8")) + return Recording( + recorded=raw["recorded"], + compiler=raw["compiler"], + target=raw["target"], + cases={ + key: Case( + name=key, + about=one["about"], + source=one["source"], + text=one["text"], + diagnostics=tuple(_stored(x) for x in one.get("diagnostics", [])), + elsewhere=one.get("elsewhere", ""), + ) + for key, one in raw["cases"].items() + }, + grammar=Grammar(functions=tuple(raw["grammar"]["functions"])), + lookahead=Lookahead( + peeks=raw["lookahead"]["peeks"], + seconds=raw["lookahead"]["seconds"], + depths={int(k): v for k, v in raw["lookahead"]["depths"].items()}, + slots=raw["lookahead"]["slots"], + ), + ) + + +def _stored(raw: dict) -> Diagnostic: + return Diagnostic( + level=raw["level"], + message=raw["message"], + at=Span(*raw["at"]), + snippet=raw.get("snippet", ""), + function=raw.get("function", ""), + fixes=tuple(Fix(at=Span(*one["at"]), insert=one["insert"]) for one in raw.get("fixes", [])), + related=tuple(Span(*one) for one in raw.get("related", [])), + ) + + +def stored(one: Diagnostic) -> dict: + """A diagnostic as the corpus keeps it, which is the inverse of `_stored`.""" + out: dict = { + "level": one.level, + "message": one.message, + "at": [one.at.line, one.at.column, one.at.end], + } + if one.snippet: + out["snippet"] = one.snippet + if one.function: + out["function"] = one.function + if one.fixes: + out["fixes"] = [ + {"at": [f.at.line, f.at.column, f.at.end], "insert": f.insert} for f in one.fixes + ] + if one.related: + out["related"] = [[s.line, s.column, s.end] for s in one.related] + return out + + +__all__ = [ + "CORPUS", + "INSERTION", + "SUFFIXES", + "CParseError", + "Case", + "Diagnostic", + "Fix", + "Grammar", + "Lookahead", + "Recording", + "Span", + "Suffix", + "load", + "parse_sarif", + "plain", + "stored", + "suffix_for", + "suffixes_for", + "undouble", +] diff --git a/gxray/glossary.py b/gxray/glossary.py index f380842..859879d 100644 --- a/gxray/glossary.py +++ b/gxray/glossary.py @@ -399,6 +399,68 @@ def anchor(name: str) -> str: ) +PARSING = Group( + "In the parser", + "The program that reads C, and the four token slots everything surprising about it comes out of. F03 is the lesson.", + ( + Term( + name="parser", + short="The recursive descent code that turns a token stream into trees. It can see four tokens.", + long="GCC's C parser is written by hand rather than generated, and all of its memory of your program is a `c_parser` struct: four token slots, a few flags, and the symbol table it shares with the rest of the front end. There is no dump flag for it, because it has no output of its own to print. What it produces is GENERIC, and what you can watch it do is complain. Nearly everything surprising about a C error message follows from the size of that buffer and from the fact that the symbol table has to answer a question before the parse can continue.", + also=("`c_parser`", "recursive descent", "`cc1`"), + cite="gcc/c/c-parser.cc:191@releases/gcc-16.2.0", + see=("lookahead", "typedef name", "GENERIC"), + met="F03", + ), + Term( + name="lookahead", + short="How far ahead the parser can look before deciding what it is reading. In C, four tokens.", + long="`c_parser_peek_token` gives the next one, `c_parser_peek_2nd_token` the one after it, and `c_parser_peek_nth_token` reaches as far as the fourth. The buffer behind all three is `c_token tokens_buf[4]` and nothing widens it. Most of the peeking in the C parser is one token deep, and the deepest constant peek in the whole file is there to recognise a version control conflict marker, which is not a C construct at all. When a grammar needs to see further than four, the parser does not get more; it commits, and then recovers.", + also=("peek", "`tokens_buf`", "LL(k)"), + cite="gcc/c/c-parser.cc:572@releases/gcc-16.2.0", + see=("parser", "token", "error recovery"), + met="F03", + ), + Term( + name="typedef name", + short="An identifier that names a type, and the reason C cannot be parsed without a symbol table.", + long="`A * b;` declares `b` as a pointer if `A` is a typedef name and multiplies two variables if it is not, and no amount of looking at tokens will tell you which. The parser asks the symbol table instead, and the answer is written into the token as `CPP_KEYWORD` or `CPP_NAME` the first time that token is looked at. That moment can come too early: a token peeked while one scope was open and used after it closed is carrying a stale answer, which is what `c_parser_maybe_reclassify_token` exists to undo.", + also=("lexer hack", "`CPP_KEYWORD`", "`c_parser_maybe_reclassify_token`"), + cite="gcc/c/c-parser.cc:2326@releases/gcc-16.2.0", + see=("parser", "token", "lookahead"), + met="F03", + ), + Term( + name="diagnostic", + short="One complaint, with a message, a severity, a place, and often a suggested repair.", + long="A diagnostic is not a line of text. It is a structure with a primary location, any number of secondary ones, and any number of fix-it hints, and the text on your terminal is one rendering of it. `-fdiagnostics-format=sarif-stderr` is another, and it is the one to reach for when you want to read the structure rather than the prose. For the parser the message is finished by `c_parse_error`, which chooses one of thirteen endings from the type of the token the parser is looking at, which is why one missing semicolon can produce eight different sentences depending on what comes after it.", + also=("`-fdiagnostics-format`", "SARIF", "`c_parse_error`"), + cite="gcc/c-family/c-common.cc:7004@releases/gcc-16.2.0", + see=("fix-it hint", "parser", "pretty printer"), + met="F03", + ), + Term( + name="fix-it hint", + short="A machine applicable edit hung off a diagnostic, saying insert or delete or replace this text here.", + long="GCC does not suggest a repair for every mistake. For a missing token it suggests one only for the seven token types `get_missing_token_insertion_kind` knows, and the hint is what decides where the caret goes: two of the seven are inserted before the token that upset the parser and five after the token before it, and in the second case the caret moves back to the end of that previous token. That is why an error about a semicolon points at the line above the one you were reading. `-fdiagnostics-parseable-fixits` prints the hints in a form an editor can apply.", + also=("`fixit_hint`", "`rich_location`", "`-fdiagnostics-parseable-fixits`"), + cite="libcpp/include/rich-location.h:620@releases/gcc-16.2.0", + see=("diagnostic", "parser"), + met="F03", + ), + Term( + name="error recovery", + short="What the parser does after an error so that it can carry on and find the next one.", + long="A parser that stopped at the first mistake would make you compile a file once per typo, so after complaining it throws tokens away until it reaches one it can start again from, usually a semicolon or a closing brace at the right nesting depth. This is why three missing semicolons can come out as two errors rather than three, and why an error near the end of a file is sometimes a consequence of one near the top rather than a mistake of its own. The habit to build is to fix the first error and compile again.", + also=("resynchronise", "`c_parser_skip_until_found`"), + cite="gcc/c/c-parser.cc:1353@releases/gcc-16.2.0", + see=("parser", "diagnostic", "lookahead"), + met="F03", + ), + ), +) + + SHAPES = Group( "The four shapes a function takes", "The same function, written down four different ways on its way to assembly. T02, T03 and T07 are the lessons.", @@ -1047,6 +1109,7 @@ def anchor(name: str) -> str: FINDING, DRIVING, PREPROCESSING, + PARSING, SHAPES, CONTROL, STATIC_SINGLE, diff --git a/lessons/CLAIMS.md b/lessons/CLAIMS.md index 98ea2bb..adad081 100644 --- a/lessons/CLAIMS.md +++ b/lessons/CLAIMS.md @@ -8,7 +8,7 @@ A few true things cannot be shown from a notebook at all: what a pass does to me Claims about GCC's source rather than its behaviour do not live here. Those carry a `path:line@tag` citation and are checked by `refcheck` against the pinned tree. -165 claims across 19 lessons, 4 of them not observable from a notebook. +172 claims across 20 lessons, 4 of them not observable from a notebook. ## C++ for people who will only ever read it @@ -269,3 +269,15 @@ Claims about GCC's source rather than its behaviour do not live here. Those carr | an include guard is recognised only if the guard is the whole file, in tokens | [`f02-68`](f02-tokens-not-text/f02.ipynb) | | GCC suggests an include guard only for a file it read exactly once | [`f02-72`](f02-tokens-not-text/f02.ipynb) | | two targets that share almost no macros expand every one of these files identically | [`f02-76`](f02-tokens-not-text/f02.ipynb) | + +## The C parser can see four tokens + +| Claim | Proved by | +| --- | --- | +| the same line of C means two different things depending on a declaration above it | [`f03-04`](f03-four-tokens/f03.ipynb) | +| swapping which of two declarations is the typedef changes whether a later line compiles | [`f03-20`](f03-four-tokens/f03.ipynb) | +| one missing semicolon produces eight different messages depending on what follows it | [`f03-26`](f03-four-tokens/f03.ipynb) | +| every parser error message ends in one of thirteen phrases chosen by token type | [`f03-34`](f03-four-tokens/f03.ipynb) | +| GCC moves the caret from the token it choked on to the place the fix-it hint goes | [`f03-44`](f03-four-tokens/f03.ipynb) | +| one diagnostic can point at three places on two lines | [`f03-58`](f03-four-tokens/f03.ipynb) | +| the same program produces character for character the same diagnostics on two targets | [`f03-70`](f03-four-tokens/f03.ipynb) | diff --git a/lessons/f03-four-tokens/build.py b/lessons/f03-four-tokens/build.py new file mode 100644 index 0000000..8898957 --- /dev/null +++ b/lessons/f03-four-tokens/build.py @@ -0,0 +1,703 @@ +"""F03. The C parser can see four tokens. + +The third lesson of Part II. F02 finished with a stream of tokens coming out of libcpp. This +is the program that reads them, and the one place in the compiler where the amount of +machinery is smaller than anybody expects. + +The lesson has one idea and everything else is a consequence of it. GCC's C parser is +recursive descent written by hand, its entire memory of your program is four token slots and +the symbol table, and nearly everything people find strange about C error messages falls out +of those two facts. Eight programs that all leave out the same semicolon get eight different +sentences. Seven of them put the caret in one place and the eighth puts it two columns later. +Three missing semicolons come out as two errors. One error points at a bracket on the line +above. Each of those is a thing to be irritated by until you know why, and then it is a thing +you can predict. + +The parser has no dump flag, because it has no output of its own; what it builds is GENERIC +and that belongs to F04. So the readout here is the diagnostic, which turns out to carry far +more than the sentence: a primary location, secondary locations, and a machine applicable +repair, all of which GCC will print as SARIF if you ask. + +Everything runs out of `corpora/diag/f03.json`: fifteen programs, twenty two diagnostics, and +two counts taken off the pinned tree. The source spans are in `corpora/source/f03.json`, so a +reader who cloned shallow and has no `vendor/gcc` sees the same parser code as everybody else. + +The reader is expected to have done F02. This lesson does not re-explain what a token is. +""" + +from tools.nbbuild import Lesson + +lesson = Lesson( + "f03-four-tokens", + "f03", + title="The C parser can see four tokens", + milestone="M3", + summary=( + "That GCC's C parser is hand written recursive descent whose whole memory is four " + "token slots and the symbol table; why one missing semicolon produces eight " + "different messages; why the caret is on the line above the mistake; why three " + "mistakes come out as two errors; and why `A * b;` needs a symbol table to read" + ), +) +badge = lesson.badge +cite = lesson.cite +term = lesson.term +claim = lesson.claim + + +lesson.md(f""" +# F03. The C parser can see four tokens + +{badge} + +Here is a line of C. + +```c +A * b; +``` + +You cannot tell what it means. If `A` is a type, it declares `b` as a pointer to `A`. If `A` +is a variable, it multiplies two numbers and throws the answer away. The tokens are identical +in both cases, so no amount of staring at them will settle it, and neither will any grammar +written over token types alone. + +GCC settles it by asking the symbol table, in the middle of lexing, before the {term("parser")} +has decided what it is reading. That one compromise is the seam this whole lesson runs along, +and the other end of it is every error message you have ever sworn at. + +You need a browser. Everything below runs on a recording of fifteen small programs, so you see +what the lesson saw. + +**What you come away with** + +- Knowing that the C parser is written by hand, is recursive descent, and can see exactly four + tokens +- Being able to explain why `A * b;` needs a {term("typedef name")} lookup, and what breaks + when the lookup happens at the wrong moment +- Being able to predict which of eight versions of the same mistake gets which sentence +- Knowing why the caret is on the line above the mistake, and when it is not +- Knowing what {term("error recovery")} is and why the first error is the only one to trust +- Being able to read a {term("diagnostic")} as a structure rather than as a line of text +- Knowing how big the C parser is, and how little of it is about C +""") + +lesson.setup() + +lesson.md(f""" +## One word, two meanings + +Start with the measurement from the top. Two programs. They differ in one word on line 1, and +line 3 is byte for byte the same in both. +{claim("the same line of C means two different things depending on a declaration above it")}. +""") + +lesson.code(""" +from gxray import cparse + +rec = cparse.load("f03") +a, b = rec["meaning-typedef"], rec["meaning-variable"] + +print(f"recorded {rec.recorded}") +print(f"{rec.compiler} for {rec.target}") +print() +for one in (a, b): + print(f"--- {one.name}: {one.about}") + print(one.source) + +print(f"line 3 is the same in both: {a.line(3) == b.line(3)}") +print(f" {a.line(3)!r}") +""") + +lesson.md(""" +Same line. Now what GCC said about each of them, with `-Wall -Wshadow` on so that it has +something to say at all. Neither program is wrong; the question is what GCC thinks line 3 is. +""") + +lesson.code(""" +for one in (a, b): + print(f"--- {one.name}") + print(one.text) +""") + +lesson.md(""" +Read the two carefully, because they are not two versions of one complaint. They are two +different readings of one line. + +In the first, `A` is a type, so `A * b;` declares a variable called `b`, and GCC warns that it +shadows the `b` at file scope and then that nobody uses it. In the second, `A` is an `int`, so +`A * b;` is a multiplication whose result goes nowhere, and GCC warns about a statement with no +effect. The caret moves too: column 20 in the first, under the `b` being declared, and column +18 in the second, spanning the whole expression with `~~^~~`. + +One word on line 1 changed which of two grammar rules line 3 matched. That is the thing C has +that most languages do not, and it is the reason the next section is about a struct rather than +about a grammar. +""") + +lesson.code(""" +for one in (a, b): + kinds = [f"{d.level}: {d.message}" for d in one.diagnostics] + print(f"{one.name:<18}{len(one.diagnostics)} diagnostics") + for text in kinds: + print(f" {text}") +""") + +lesson.md(f""" +## Four slots + +The C parser is not generated. There is no grammar file, no table, and nothing that looks like +yacc. It is a few hundred functions that call each other, one per construct in the language, +and its entire state is one struct. Here is the top of it. +""") + +lesson.code(""" +from gxray import source + +cuts = source.load_extract("f03") +row = cuts["slots"] + +print(f"{row.span} ({row.citation})") +print(row.about) +print() +print(row.numbered()) +""") + +lesson.md(f""" +`c_token tokens_buf[4]`. Four. That is how much of your program the parser can see at once, and +the comment above the struct says two, which was true when somebody wrote it and has not been +true for years. The buffer is the whole of the {term("lookahead")}, and nothing anywhere widens +it. + +Note what is not in that struct: your file. There is no buffer of text, no array of every +token, no tree of what has been read so far. Tokens arrive one at a time, are looked at, and +are gone. The C++ front end does lex the whole file up front, and the C front end has a comment +about wishing it did. +""") + +lesson.code(""" +print(cuts["intermediates"].numbered()) +""") + +lesson.md(f""" +Two more things worth having in your head before the evidence starts. + +The first is that everything the parser will ever know about your program arrives through one +function. This is the seam F02 finished at, seen from the other side. +""") + +lesson.code(""" +print(cuts["handover"].numbered()) +""") + +lesson.md(""" +`c_lex_with_flags`. One token, out of libcpp, into a `c_token`. Every fact in this lesson is +downstream of that call, and the parser has no other way of finding anything out. + +The second is that the top level of the C language, the rule that says what a whole file is, is +six lines of code. +""") + +lesson.code(""" +print(cuts["loop"].numbered()) +""") + +lesson.md(f""" +A `do` loop calling `c_parser_external_declaration` until the next token is the end of the file, +with a garbage collection on each turn. A {term("translation unit")} is a list of declarations, +and the code says so about as directly as code can. + +## The symbol table answers early + +Back to `A * b;`. Somebody has to decide whether `A` is a type, and the place it happens is not +where you would guess. It is in the lexer, while the token is being made. +""") + +lesson.code(""" +print(cuts["lookup"].numbered()) +""") + +lesson.md(f""" +`lookup_name`, then `TYPE_DECL`, then the answer written into the token as `C_ID_TYPENAME` or +`C_ID_ID`. The token stops being an identifier and becomes an identifier-that-names-a-type, +permanently, at the moment it is first lexed. + +Permanently is the problem. A token can be peeked while one scope is open and used after that +scope has closed, and then it is carrying an answer from a symbol table that no longer looks +like that. Here is the smallest program anybody found that does it. +{claim("swapping which of two declarations is the typedef changes whether a later line compiles")}. +""") + +lesson.code(""" +first, second = rec["scope-typedef"], rec["scope-variable"] + +for one in (first, second): + print(f"--- {one.name}: {one.about}") + print(one.source) +""") + +lesson.md(""" +Line 4 opens a `for` header, which is a scope. It declares `T` inside that scope, one way in +each program. Line 7, after the loop has closed, uses `T`. Predict what happens to line 7 in +each before running the next cell. +""") + +lesson.code(""" +for one in (first, second): + print(f"--- {one.name}") + print(one.text or "nothing at all") +""") + +lesson.md(f""" +In the first, `T` is a file scope type shadowed by a local variable, the shadow ends with the +loop, and `T *x;` on line 7 declares a pointer. Two unused variable warnings and no errors. + +In the second the two are swapped, so `T` is a file scope variable shadowed by a local typedef, +and by line 7 the typedef is out of scope, so `T *x;` is a multiplication of `T` by an undeclared +`x`. GCC says `'x' undeclared`, which is a strange thing to be told about a line that looks like +a declaration, and is exactly right. + +The reason the first one works is a function that exists to undo the early answer. +""") + +lesson.code(""" +print(cuts["reclassify"].numbered()) +""") + +lesson.md(f""" +`c_parser_maybe_reclassify_token`, and the comment names the bug: PR67784, a `for` loop with an +`if` and no `else` in the body. Both of the programs above are distilled from that bug's test +cases, which are in the tree at `gcc/testsuite/gcc.dg/pr67784-1.c` and `-2.c`. The fix is to +look the identifier up again, after the scope closes, and overwrite the answer. + +Two things to take from this. One is that the lexer hack is not a metaphor: the classification +is literally stored in the token. The other is that when a design has a seam in it, the bugs +gather at the seam, and the patches are named after the bug reports. + +## One mistake, eight sentences + +Now the part of the lesson that is worth the price of admission. + +Eight programs. Every one of them is the same mistake: a missing semicolon after `return 1`. +They differ in what comes next, and nothing else. +{claim("one missing semicolon produces eight different messages depending on what follows it")}. +""") + +lesson.code(""" +same = ("brace", "name", "number", "string", "char", "keyword", "pragma", "eof") + +for key in same: + one = rec[key] + print(f"{key:<10}{one.line(1)!r}") +""") + +lesson.md(""" +Eight programs, one mistake. Now the messages. Read the list before reading the paragraph after +it, because the shape of the list is the point. +""") + +lesson.code(""" +for key in same: + error = rec[key].errors[0] + print(f"{key:<10}{str(error.at):<10}{error.message}") + +sentences = {rec[k].errors[0].message for k in same} +print() +print(f"{len(sentences)} different sentences from {len(same)} copies of one mistake") +""") + +lesson.md(f""" +Eight sentences. The parser wanted a semicolon in every one of them and the semicolon was +missing in every one of them, and it said eight different things, because the second half of +the sentence is chosen by the token it happened to be looking at when it gave up. + +Look at the column too. Seven of the eight say 23. One says 25. Hold on to that; it is the next +section but one. + +## Where the sentence is made + +There is one function that finishes these sentences, and it is shared by the C and C++ front +ends. A parser function passes in a complaint like `expected ';'` and the token type, and gets +back a whole sentence. +""") + +lesson.code(""" +print(cuts["suffixes"].numbered()) +""") + +lesson.md(f""" +`catenate_messages (gmsgid, " at end of input")`. That is the first of thirteen branches, and +the rest are the same shape. `gxray` has all thirteen transcribed, so you can see the range of +endings a complaint can be given. +""") + +lesson.code(""" +for one in cparse.SUFFIXES: + kinds = ", ".join(one.types) or "everything else" + print(f"{one.text!r}") + print(f" {one.about}") + print(f" {kinds}") +""") + +lesson.md(f""" +Thirteen endings, and the last is a catch-all with a range test rather than a list, which is +where every punctuation mark in the language ends up. That is why `expected ';' before '}}' +token` has the word `token` on the end and `expected ';' before numeric constant` does not: two +different branches wrote them. + +Now the check. Take the eight recorded messages and ask which branch could have produced each. +{claim("every parser error message ends in one of thirteen phrases chosen by token type")}. +""") + +lesson.code(""" +for key in same: + error = rec[key].errors[0] + found = error.suffixes + which = found[0].about if len(found) == 1 else f"cannot tell: {len(found)} branches match" + print(f"{key:<10}{error.message:<42}{which}") +""") + +lesson.md(""" +Six of the eight are pinned to one branch. Two are not, and the reason is worth a moment. + +`char` and `name` both end in a single character inside single quotes. One of them is a +character constant, printed by the branch for character constants. The other is a one letter +identifier, printed by the branch that quotes identifiers back at you with `%qE`. Rendered as +text they are indistinguishable, and no amount of care in this module can separate them. + +That is a real limit and it is worth naming rather than papering over. The recorded message is +all a notebook has. GCC, which had the token, knew perfectly well which one it was. + +## Where the caret goes + +Back to the column. Seven of the eight put the caret at 23 and one puts it at 25. +""") + +lesson.code(""" +for key in same: + error = rec[key].errors[0] + hint = f"fix-it: insert {error.fixes[0].insert!r}" if error.fixes else "no fix-it" + print(f"{key:<10}column {error.at.column:<4}{hint}") +""") + +lesson.md(f""" +The split is exact. Seven carry a {term("fix-it hint")} and sit at column 23. One has no hint +and sits at column 25. Here is the odd one out with its caret, next to one of the seven. +""") + +lesson.code(""" +for key in ("brace", "name"): + print(f"--- {key}") + print(rec[key].text) +""") + +lesson.md(f""" +Column 23 is one past the `1` in `return 1`, which is where the semicolon should have gone. +Column 25 is the `b`, which is the token that upset the parser. So the seven point at the +mistake and the odd one points at the symptom, and the difference between them is whether GCC +was willing to suggest a repair. + +It will suggest one for seven token types, and no others. +""") + +lesson.code(""" +print(cuts["insertion"].numbered()) +""") + +lesson.md(f""" +Two go before the next token, five go after the previous one, and everything else gets nothing. +`gxray` keeps the same table, so a notebook can talk about it. +""") + +lesson.code(""" +for token, side in cparse.INSERTION.items(): + print(f" {token!r:<6}goes {side}") + +print() +print(f"{sum(1 for s in cparse.INSERTION.values() if s == 'before')} before the next token") +print(f"{sum(1 for s in cparse.INSERTION.values() if s == 'after')} after the previous one") +""") + +lesson.md(f""" +And then the part that actually moves the caret. When GCC works out where the missing token +should go, it decides that the repair is a better place to point than the token that tripped +over. So it swaps them: the fix-it location becomes the primary one and the old primary becomes +a secondary. GCC explains this in its own comment, with a diagram. +{claim("GCC moves the caret from the token it choked on to the place the fix-it hint goes")}. +""") + +lesson.code(""" +print(cuts["swap"].numbered()) +""") + +lesson.md(""" +That comment is the single most useful thing in this lesson to have read once. Every time an +error about a semicolon points at the line above the one you were editing, this is why, and it +is deliberate, and it is right. + +The swap leaves evidence. The old primary location is still on the diagnostic, as a secondary, +and the recording keeps those. +""") + +lesson.code(""" +one = rec["brace"] +error = one.errors[0] + +print(f"caret at {error.at}") +print(f"fix-it at {error.fixes[0].at}, inserting {error.fixes[0].insert!r}") +print(f"secondary {', '.join(str(s) for s in error.related)}") +print(f"caret moved {error.moved}") +print() +print(f" {one.line(1)}") +print(f" {one.under(error)} the caret, at the missing semicolon") +""") + +lesson.md(f""" +The secondary is at 1:24, which is the `}}`. The caret is at 1:23, which is the gap. Both are +on the diagnostic and only one of them is drawn with a `^`. + +Two functions produce parser errors and the difference between them is exactly this. One puts +the caret on the token it can see. +""") + +lesson.code(""" +print(cuts["report"].numbered()) +""") + +lesson.md(f""" +The other knows which token you left out, and so can offer a hint and move the caret. +""") + +lesson.code(""" +print(cuts["require"].numbered()) +""") + +lesson.md(""" +`maybe_suggest_missing_token_insertion`, then `add_location_if_nearby`, then the error. The +`name` case went through the first function, because `expected ',' or ';'` names two possible +tokens and `type_is_unique` is false, and you cannot suggest inserting one of two things. So no +hint, no swap, and the caret stays on the `b`. + +One error message, one extra token in the program, and a whole different code path. That is +what the column 23 against column 25 split was telling you. + +## Three mistakes, two errors + +Next irritation. Here is a function with three missing semicolons in it. +{claim("three missing semicolons produce two errors, not three")}. +""") + +lesson.code(""" +one = rec["recovery"] + +print(one.source) +print(one.text) +print(f"{len(one.errors)} errors for three mistakes") +""") + +lesson.md(f""" +Two errors for three mistakes, and the second one is not about a semicolon at all. It says +`expected declaration or statement at end of input`, pointing at the closing brace on line 6, +which is a fine and correct closing brace. + +Two things happened. The first is {term("error recovery")}: after complaining on line 4 the +parser threw tokens away looking for somewhere to start again, and what it found was the end of +the function. The second is a latch. +""") + +lesson.code(""" +print(cuts["latch"].numbered()) +""") + +lesson.md(f""" +`if (parser->error) return false;` and then `parser->error = true;`. Once the parser has +complained, it will not complain again until something clears the flag, which is what stops one +missing brace from producing four hundred errors. + +The rule to carry away is short. **The first error is the one to trust.** Everything after it +was produced by a parser that had already lost its place, and the usual outcome of fixing the +first one is that the rest go away. + +## The bracket it points back at + +One more piece of the diagnostic structure, because it is the one people notice and cannot +explain. Here is an unclosed bracket. +""") + +lesson.code(""" +one = rec["paren"] + +print(one.source) +print(one.text) +""") + +lesson.md(f""" +Look at what that drew. A caret on line 4, a `~` under the `(` on line 4, and another `~` under +the `g` on line 5. Three places, one message, two lines apart. +{claim("one diagnostic can point at three places on two lines")}. +""") + +lesson.code(""" +error = one.errors[0] + +print(f"caret {error.at}") +print(f"fix-it {error.fixes[0].at}, inserting {error.fixes[0].insert!r}") +for span in error.related: + print(f"secondary {span} {one.line(span.line)!r}") +""") + +lesson.md(""" +That is `matching_location` in `c_parser_require`, the argument the last section walked past. +When a parser function asks for a closing bracket it passes the location of the opening one, and +`add_location_if_nearby` folds it into the same diagnostic if it will fit on the display. + +Which is why the message is `expected ')' before 'g'` and not `expected ')'`: three locations, +one sentence, and enough for you to see the bracket you left open without going to look for it. + +## The deepest it ever looks + +The parser has four slots. The obvious question is what needs the fourth, and the answer is not +a C construct. +""") + +lesson.code(""" +one = rec["conflict"] + +print(one.source) +print(one.text) +""") + +lesson.md(f""" +Three errors, all the same sentence, each one seven columns wide. That is not a parse error. It +is GCC recognising that somebody committed a merge conflict. +""") + +lesson.code(""" +print(cuts["conflict"].numbered()) +""") + +lesson.md(f""" +`c_parser_peek_2nd_token`, then `peek_nth_token (parser, 3)`, then `peek_nth_token (parser, 4)`. +A run of seven `<` characters arrives from the lexer as three `CPP_LSHIFT` tokens and one +`CPP_LESS`, so recognising it takes four tokens, which is exactly how many there are. Then a +column check, because a conflict marker is at the start of a line and `a << b << c << d` is not. + +Count how often the parser looks that far. +""") + +lesson.code(""" +look = rec.lookahead + +print(f"{look.slots} token slots in the struct") +print(f"{look.peeks} calls to c_parser_peek_token, which looks at one") +print(f"{look.seconds} calls to c_parser_peek_2nd_token, which looks at two") +print() +for depth in sorted(look.depths): + print(f"{look.depths[depth]:>3} calls to peek_nth_token with a constant {depth}") +print(f"deepest constant peek: {look.deepest}") +""") + +lesson.md(""" +Seven hundred and twenty nine one token peeks, ninety nine two token peeks, and eleven that name +a depth as a constant. Three of those eleven are the four deep ones, and all three are in the +conflict marker function you have on the screen. + +So the fourth slot exists for a thing that is not part of C, and the entire C language is parsed +in three. That is not a criticism of anybody. It is what a language designed to be compiled in +one pass on a PDP-11 looks like from the inside, and it is the reason the error messages are the +shape they are: a parser that can see three tokens cannot know what you meant. + +## The size of the thing + +Two counts to finish, both taken off the pinned tree rather than described. +""") + +lesson.code(""" +grammar = rec.grammar +parts = grammar.dialects + +print(f"{len(grammar)} functions named c_parser_* in gcc/c/c-parser.cc") +print() +for name, names in sorted(parts.items(), key=lambda kv: -len(kv[1])): + share = 100 * len(names) / len(grammar) + print(f"{name:<22}{len(names):>4} {share:4.1f}%") +""") + +lesson.md(f""" +Under half the C parser is for C. There are more functions for OpenMP than for the language, and +the total is three hundred once you add OpenACC, Objective-C and transactional memory. If you go +looking for the function that parses a `while` loop and find yourself in a file whose contents +are mostly pragma directives for a parallelism standard, that is why. + +Here is a sample of the ones that are about C, so the naming convention is visible. +""") + +lesson.code(""" +wanted = ("declaration", "statement", "expression", "initializer", "struct", "label", "typeof") + +for name in grammar.named(*wanted): + print(f" {name}") +""") + +lesson.md(f""" +`c_parser_` and then the name of the thing in the grammar. Once you know that, finding the code +for any construct in C is one search, which is the practical thing this section is for. + +## Same parser, different machine + +One last check, and it is the one that says which of the facts above are about C and which are +about this laptop. Three of the programs were also compiled by an x86-64 Linux GCC of the same +release, through Compiler Explorer. +{claim("the same program produces character for character the same diagnostics on two targets")}. +""") + +lesson.code(""" +shared = [one for one in rec if one.elsewhere] + +for one in shared: + print(f"{one.name:<20}{'same' if one.agrees else 'DIFFERENT'} {one.about}") + +print() +print(f"{len(shared)} programs compiled twice, on {rec.target} and on x86-64 Linux") +""") + +lesson.md(f""" +Identical. Which is the shape of the subject, the same as it was in F02: whether `A * b;` is a +declaration is a fact about C, and it will read the same on a compiler for a processor nobody +has built yet. The target gets a say in what `int` is worth. It gets no say at all in what the +parser thinks you wrote. + +## Boss fight + +No new tools. Three questions, answerable from the recording: + +1. Eight programs, one missing semicolon each, eight different sentences. Seven of them put the + caret at column 23. Which one does not, and what is different about it? +2. Thirteen phrases can end a parser complaint, and two of the eight recorded messages cannot be + pinned to one of them from the text alone. Which two, and why not? +3. The parser has four token slots. How many calls in the whole file pass a constant 4 to + `c_parser_peek_nth_token`, and what are they all for? + +Then check yourself: + +```text +python lessons/f03-four-tokens/grade.py +``` + +or, from a checkout, `just grade f03-four-tokens`. It takes answers on the command line too, so +`--odd name --unsure char,name --deep 3` is a whole submission. Every answer it marks against is +worked out from the recording rather than written down. + +## What to read next + +F04 is GENERIC, which is what this parser produces. The parser has been throwing tokens away +this whole lesson and building something out of them, and F04 is the something. + +F02 is the lesson before this one, if you have not done it. It is where the tokens come from. + +If you have a GCC on your machine, every measurement here is one command. +`printf 'int f(void) {{ return 1 }}' | gcc -fsyntax-only -xc -` is the first one, and +`-fdiagnostics-format=sarif-stderr` on the end of it is the structure behind the sentence. Change +the `}}` to a `1` and watch the column move. +""") + +raise SystemExit(lesson.save()) diff --git a/lessons/f03-four-tokens/f03.ipynb b/lessons/f03-four-tokens/f03.ipynb new file mode 100644 index 0000000..24e60d9 --- /dev/null +++ b/lessons/f03-four-tokens/f03.ipynb @@ -0,0 +1,1053 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "id": "f03-01", + "metadata": {}, + "source": [ + "# F03. The C parser can see four tokens\n", + "\n", + "[![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/tamnd/gcc-internals/blob/main/lessons/f03-four-tokens/f03.ipynb)\n", + "\n", + "Here is a line of C.\n", + "\n", + "```c\n", + "A * b;\n", + "```\n", + "\n", + "You cannot tell what it means. If `A` is a type, it declares `b` as a pointer to `A`. If `A`\n", + "is a variable, it multiplies two numbers and throws the answer away. The tokens are identical\n", + "in both cases, so no amount of staring at them will settle it, and neither will any grammar\n", + "written over token types alone.\n", + "\n", + "GCC settles it by asking the symbol table, in the middle of lexing, before the [parser](https://github.com/tamnd/gcc-internals/blob/main/GLOSSARY.md#parser)\n", + "has decided what it is reading. That one compromise is the seam this whole lesson runs along,\n", + "and the other end of it is every error message you have ever sworn at.\n", + "\n", + "You need a browser. Everything below runs on a recording of fifteen small programs, so you see\n", + "what the lesson saw.\n", + "\n", + "**What you come away with**\n", + "\n", + "- Knowing that the C parser is written by hand, is recursive descent, and can see exactly four\n", + " tokens\n", + "- Being able to explain why `A * b;` needs a [typedef name](https://github.com/tamnd/gcc-internals/blob/main/GLOSSARY.md#typedef-name) lookup, and what breaks\n", + " when the lookup happens at the wrong moment\n", + "- Being able to predict which of eight versions of the same mistake gets which sentence\n", + "- Knowing why the caret is on the line above the mistake, and when it is not\n", + "- Knowing what [error recovery](https://github.com/tamnd/gcc-internals/blob/main/GLOSSARY.md#error-recovery) is and why the first error is the only one to trust\n", + "- Being able to read a [diagnostic](https://github.com/tamnd/gcc-internals/blob/main/GLOSSARY.md#diagnostic) as a structure rather than as a line of text\n", + "- Knowing how big the C parser is, and how little of it is about C" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-02", + "metadata": {}, + "outputs": [], + "source": [ + "# Setup. Safe to run twice, and does nothing at all if you already have the repository.\n", + "#\n", + "# Colab starts with none of this project's files and a GCC that is several years older than\n", + "# the one the lessons were written against. This cell fixes the first problem by cloning,\n", + "# and works around the second by having gxray fall back to recorded dumps or to Compiler\n", + "# Explorer, both of which are really GCC 16.\n", + "import os\n", + "import subprocess\n", + "import sys\n", + "\n", + "REPO = \"https://github.com/tamnd/gcc-internals\"\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " if not os.path.isdir(\"gcc-internals\"):\n", + " subprocess.run([\"git\", \"clone\", \"--depth\", \"1\", \"-q\", REPO], check=True)\n", + " if os.path.basename(os.getcwd()) != \"gcc-internals\":\n", + " os.chdir(\"gcc-internals\")\n", + " if os.getcwd() not in sys.path:\n", + " sys.path.insert(0, os.getcwd())\n", + "\n", + "import gxray\n", + "\n", + "print(\"gxray\", gxray.__version__, \"on python\", sys.version.split()[0])" + ] + }, + { + "cell_type": "markdown", + "id": "f03-03", + "metadata": {}, + "source": [ + "## One word, two meanings\n", + "\n", + "Start with the measurement from the top. Two programs. They differ in one word on line 1, and\n", + "line 3 is byte for byte the same in both.\n", + "the same line of C means two different things depending on a declaration above it." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-04", + "metadata": {}, + "outputs": [], + "source": [ + "from gxray import cparse\n", + "\n", + "rec = cparse.load(\"f03\")\n", + "a, b = rec[\"meaning-typedef\"], rec[\"meaning-variable\"]\n", + "\n", + "print(f\"recorded {rec.recorded}\")\n", + "print(f\"{rec.compiler} for {rec.target}\")\n", + "print()\n", + "for one in (a, b):\n", + " print(f\"--- {one.name}: {one.about}\")\n", + " print(one.source)\n", + "\n", + "print(f\"line 3 is the same in both: {a.line(3) == b.line(3)}\")\n", + "print(f\" {a.line(3)!r}\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-05", + "metadata": {}, + "source": [ + "Same line. Now what GCC said about each of them, with `-Wall -Wshadow` on so that it has\n", + "something to say at all. Neither program is wrong; the question is what GCC thinks line 3 is." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-06", + "metadata": {}, + "outputs": [], + "source": [ + "for one in (a, b):\n", + " print(f\"--- {one.name}\")\n", + " print(one.text)" + ] + }, + { + "cell_type": "markdown", + "id": "f03-07", + "metadata": {}, + "source": [ + "Read the two carefully, because they are not two versions of one complaint. They are two\n", + "different readings of one line.\n", + "\n", + "In the first, `A` is a type, so `A * b;` declares a variable called `b`, and GCC warns that it\n", + "shadows the `b` at file scope and then that nobody uses it. In the second, `A` is an `int`, so\n", + "`A * b;` is a multiplication whose result goes nowhere, and GCC warns about a statement with no\n", + "effect. The caret moves too: column 20 in the first, under the `b` being declared, and column\n", + "18 in the second, spanning the whole expression with `~~^~~`.\n", + "\n", + "One word on line 1 changed which of two grammar rules line 3 matched. That is the thing C has\n", + "that most languages do not, and it is the reason the next section is about a struct rather than\n", + "about a grammar." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-08", + "metadata": {}, + "outputs": [], + "source": [ + "for one in (a, b):\n", + " kinds = [f\"{d.level}: {d.message}\" for d in one.diagnostics]\n", + " print(f\"{one.name:<18}{len(one.diagnostics)} diagnostics\")\n", + " for text in kinds:\n", + " print(f\" {text}\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-09", + "metadata": {}, + "source": [ + "## Four slots\n", + "\n", + "The C parser is not generated. There is no grammar file, no table, and nothing that looks like\n", + "yacc. It is a few hundred functions that call each other, one per construct in the language,\n", + "and its entire state is one struct. Here is the top of it." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-10", + "metadata": {}, + "outputs": [], + "source": [ + "from gxray import source\n", + "\n", + "cuts = source.load_extract(\"f03\")\n", + "row = cuts[\"slots\"]\n", + "\n", + "print(f\"{row.span} ({row.citation})\")\n", + "print(row.about)\n", + "print()\n", + "print(row.numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-11", + "metadata": {}, + "source": [ + "`c_token tokens_buf[4]`. Four. That is how much of your program the parser can see at once, and\n", + "the comment above the struct says two, which was true when somebody wrote it and has not been\n", + "true for years. The buffer is the whole of the [lookahead](https://github.com/tamnd/gcc-internals/blob/main/GLOSSARY.md#lookahead), and nothing anywhere widens\n", + "it.\n", + "\n", + "Note what is not in that struct: your file. There is no buffer of text, no array of every\n", + "token, no tree of what has been read so far. Tokens arrive one at a time, are looked at, and\n", + "are gone. The C++ front end does lex the whole file up front, and the C front end has a comment\n", + "about wishing it did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-12", + "metadata": {}, + "outputs": [], + "source": [ + "print(cuts[\"intermediates\"].numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-13", + "metadata": {}, + "source": [ + "Two more things worth having in your head before the evidence starts.\n", + "\n", + "The first is that everything the parser will ever know about your program arrives through one\n", + "function. This is the seam F02 finished at, seen from the other side." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-14", + "metadata": {}, + "outputs": [], + "source": [ + "print(cuts[\"handover\"].numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-15", + "metadata": {}, + "source": [ + "`c_lex_with_flags`. One token, out of libcpp, into a `c_token`. Every fact in this lesson is\n", + "downstream of that call, and the parser has no other way of finding anything out.\n", + "\n", + "The second is that the top level of the C language, the rule that says what a whole file is, is\n", + "six lines of code." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-16", + "metadata": {}, + "outputs": [], + "source": [ + "print(cuts[\"loop\"].numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-17", + "metadata": {}, + "source": [ + "A `do` loop calling `c_parser_external_declaration` until the next token is the end of the file,\n", + "with a garbage collection on each turn. A [translation unit](https://github.com/tamnd/gcc-internals/blob/main/GLOSSARY.md#translation-unit) is a list of declarations,\n", + "and the code says so about as directly as code can.\n", + "\n", + "## The symbol table answers early\n", + "\n", + "Back to `A * b;`. Somebody has to decide whether `A` is a type, and the place it happens is not\n", + "where you would guess. It is in the lexer, while the token is being made." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-18", + "metadata": {}, + "outputs": [], + "source": [ + "print(cuts[\"lookup\"].numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-19", + "metadata": {}, + "source": [ + "`lookup_name`, then `TYPE_DECL`, then the answer written into the token as `C_ID_TYPENAME` or\n", + "`C_ID_ID`. The token stops being an identifier and becomes an identifier-that-names-a-type,\n", + "permanently, at the moment it is first lexed.\n", + "\n", + "Permanently is the problem. A token can be peeked while one scope is open and used after that\n", + "scope has closed, and then it is carrying an answer from a symbol table that no longer looks\n", + "like that. Here is the smallest program anybody found that does it.\n", + "swapping which of two declarations is the typedef changes whether a later line compiles." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-20", + "metadata": {}, + "outputs": [], + "source": [ + "first, second = rec[\"scope-typedef\"], rec[\"scope-variable\"]\n", + "\n", + "for one in (first, second):\n", + " print(f\"--- {one.name}: {one.about}\")\n", + " print(one.source)" + ] + }, + { + "cell_type": "markdown", + "id": "f03-21", + "metadata": {}, + "source": [ + "Line 4 opens a `for` header, which is a scope. It declares `T` inside that scope, one way in\n", + "each program. Line 7, after the loop has closed, uses `T`. Predict what happens to line 7 in\n", + "each before running the next cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-22", + "metadata": {}, + "outputs": [], + "source": [ + "for one in (first, second):\n", + " print(f\"--- {one.name}\")\n", + " print(one.text or \"nothing at all\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-23", + "metadata": {}, + "source": [ + "In the first, `T` is a file scope type shadowed by a local variable, the shadow ends with the\n", + "loop, and `T *x;` on line 7 declares a pointer. Two unused variable warnings and no errors.\n", + "\n", + "In the second the two are swapped, so `T` is a file scope variable shadowed by a local typedef,\n", + "and by line 7 the typedef is out of scope, so `T *x;` is a multiplication of `T` by an undeclared\n", + "`x`. GCC says `'x' undeclared`, which is a strange thing to be told about a line that looks like\n", + "a declaration, and is exactly right.\n", + "\n", + "The reason the first one works is a function that exists to undo the early answer." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-24", + "metadata": {}, + "outputs": [], + "source": [ + "print(cuts[\"reclassify\"].numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-25", + "metadata": {}, + "source": [ + "`c_parser_maybe_reclassify_token`, and the comment names the bug: PR67784, a `for` loop with an\n", + "`if` and no `else` in the body. Both of the programs above are distilled from that bug's test\n", + "cases, which are in the tree at `gcc/testsuite/gcc.dg/pr67784-1.c` and `-2.c`. The fix is to\n", + "look the identifier up again, after the scope closes, and overwrite the answer.\n", + "\n", + "Two things to take from this. One is that the lexer hack is not a metaphor: the classification\n", + "is literally stored in the token. The other is that when a design has a seam in it, the bugs\n", + "gather at the seam, and the patches are named after the bug reports.\n", + "\n", + "## One mistake, eight sentences\n", + "\n", + "Now the part of the lesson that is worth the price of admission.\n", + "\n", + "Eight programs. Every one of them is the same mistake: a missing semicolon after `return 1`.\n", + "They differ in what comes next, and nothing else.\n", + "one missing semicolon produces eight different messages depending on what follows it." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-26", + "metadata": {}, + "outputs": [], + "source": [ + "same = (\"brace\", \"name\", \"number\", \"string\", \"char\", \"keyword\", \"pragma\", \"eof\")\n", + "\n", + "for key in same:\n", + " one = rec[key]\n", + " print(f\"{key:<10}{one.line(1)!r}\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-27", + "metadata": {}, + "source": [ + "Eight programs, one mistake. Now the messages. Read the list before reading the paragraph after\n", + "it, because the shape of the list is the point." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-28", + "metadata": {}, + "outputs": [], + "source": [ + "for key in same:\n", + " error = rec[key].errors[0]\n", + " print(f\"{key:<10}{str(error.at):<10}{error.message}\")\n", + "\n", + "sentences = {rec[k].errors[0].message for k in same}\n", + "print()\n", + "print(f\"{len(sentences)} different sentences from {len(same)} copies of one mistake\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-29", + "metadata": {}, + "source": [ + "Eight sentences. The parser wanted a semicolon in every one of them and the semicolon was\n", + "missing in every one of them, and it said eight different things, because the second half of\n", + "the sentence is chosen by the token it happened to be looking at when it gave up.\n", + "\n", + "Look at the column too. Seven of the eight say 23. One says 25. Hold on to that; it is the next\n", + "section but one.\n", + "\n", + "## Where the sentence is made\n", + "\n", + "There is one function that finishes these sentences, and it is shared by the C and C++ front\n", + "ends. A parser function passes in a complaint like `expected ';'` and the token type, and gets\n", + "back a whole sentence." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-30", + "metadata": {}, + "outputs": [], + "source": [ + "print(cuts[\"suffixes\"].numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-31", + "metadata": {}, + "source": [ + "`catenate_messages (gmsgid, \" at end of input\")`. That is the first of thirteen branches, and\n", + "the rest are the same shape. `gxray` has all thirteen transcribed, so you can see the range of\n", + "endings a complaint can be given." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-32", + "metadata": {}, + "outputs": [], + "source": [ + "for one in cparse.SUFFIXES:\n", + " kinds = \", \".join(one.types) or \"everything else\"\n", + " print(f\"{one.text!r}\")\n", + " print(f\" {one.about}\")\n", + " print(f\" {kinds}\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-33", + "metadata": {}, + "source": [ + "Thirteen endings, and the last is a catch-all with a range test rather than a list, which is\n", + "where every punctuation mark in the language ends up. That is why `expected ';' before '}'\n", + "token` has the word `token` on the end and `expected ';' before numeric constant` does not: two\n", + "different branches wrote them.\n", + "\n", + "Now the check. Take the eight recorded messages and ask which branch could have produced each.\n", + "every parser error message ends in one of thirteen phrases chosen by token type." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-34", + "metadata": {}, + "outputs": [], + "source": [ + "for key in same:\n", + " error = rec[key].errors[0]\n", + " found = error.suffixes\n", + " which = found[0].about if len(found) == 1 else f\"cannot tell: {len(found)} branches match\"\n", + " print(f\"{key:<10}{error.message:<42}{which}\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-35", + "metadata": {}, + "source": [ + "Six of the eight are pinned to one branch. Two are not, and the reason is worth a moment.\n", + "\n", + "`char` and `name` both end in a single character inside single quotes. One of them is a\n", + "character constant, printed by the branch for character constants. The other is a one letter\n", + "identifier, printed by the branch that quotes identifiers back at you with `%qE`. Rendered as\n", + "text they are indistinguishable, and no amount of care in this module can separate them.\n", + "\n", + "That is a real limit and it is worth naming rather than papering over. The recorded message is\n", + "all a notebook has. GCC, which had the token, knew perfectly well which one it was.\n", + "\n", + "## Where the caret goes\n", + "\n", + "Back to the column. Seven of the eight put the caret at 23 and one puts it at 25." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-36", + "metadata": {}, + "outputs": [], + "source": [ + "for key in same:\n", + " error = rec[key].errors[0]\n", + " hint = f\"fix-it: insert {error.fixes[0].insert!r}\" if error.fixes else \"no fix-it\"\n", + " print(f\"{key:<10}column {error.at.column:<4}{hint}\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-37", + "metadata": {}, + "source": [ + "The split is exact. Seven carry a [fix-it hint](https://github.com/tamnd/gcc-internals/blob/main/GLOSSARY.md#fix-it-hint) and sit at column 23. One has no hint\n", + "and sits at column 25. Here is the odd one out with its caret, next to one of the seven." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-38", + "metadata": {}, + "outputs": [], + "source": [ + "for key in (\"brace\", \"name\"):\n", + " print(f\"--- {key}\")\n", + " print(rec[key].text)" + ] + }, + { + "cell_type": "markdown", + "id": "f03-39", + "metadata": {}, + "source": [ + "Column 23 is one past the `1` in `return 1`, which is where the semicolon should have gone.\n", + "Column 25 is the `b`, which is the token that upset the parser. So the seven point at the\n", + "mistake and the odd one points at the symptom, and the difference between them is whether GCC\n", + "was willing to suggest a repair.\n", + "\n", + "It will suggest one for seven token types, and no others." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-40", + "metadata": {}, + "outputs": [], + "source": [ + "print(cuts[\"insertion\"].numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-41", + "metadata": {}, + "source": [ + "Two go before the next token, five go after the previous one, and everything else gets nothing.\n", + "`gxray` keeps the same table, so a notebook can talk about it." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-42", + "metadata": {}, + "outputs": [], + "source": [ + "for token, side in cparse.INSERTION.items():\n", + " print(f\" {token!r:<6}goes {side}\")\n", + "\n", + "print()\n", + "print(f\"{sum(1 for s in cparse.INSERTION.values() if s == 'before')} before the next token\")\n", + "print(f\"{sum(1 for s in cparse.INSERTION.values() if s == 'after')} after the previous one\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-43", + "metadata": {}, + "source": [ + "And then the part that actually moves the caret. When GCC works out where the missing token\n", + "should go, it decides that the repair is a better place to point than the token that tripped\n", + "over. So it swaps them: the fix-it location becomes the primary one and the old primary becomes\n", + "a secondary. GCC explains this in its own comment, with a diagram.\n", + "GCC moves the caret from the token it choked on to the place the fix-it hint goes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-44", + "metadata": {}, + "outputs": [], + "source": [ + "print(cuts[\"swap\"].numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-45", + "metadata": {}, + "source": [ + "That comment is the single most useful thing in this lesson to have read once. Every time an\n", + "error about a semicolon points at the line above the one you were editing, this is why, and it\n", + "is deliberate, and it is right.\n", + "\n", + "The swap leaves evidence. The old primary location is still on the diagnostic, as a secondary,\n", + "and the recording keeps those." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-46", + "metadata": {}, + "outputs": [], + "source": [ + "one = rec[\"brace\"]\n", + "error = one.errors[0]\n", + "\n", + "print(f\"caret at {error.at}\")\n", + "print(f\"fix-it at {error.fixes[0].at}, inserting {error.fixes[0].insert!r}\")\n", + "print(f\"secondary {', '.join(str(s) for s in error.related)}\")\n", + "print(f\"caret moved {error.moved}\")\n", + "print()\n", + "print(f\" {one.line(1)}\")\n", + "print(f\" {one.under(error)} the caret, at the missing semicolon\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-47", + "metadata": {}, + "source": [ + "The secondary is at 1:24, which is the `}`. The caret is at 1:23, which is the gap. Both are\n", + "on the diagnostic and only one of them is drawn with a `^`.\n", + "\n", + "Two functions produce parser errors and the difference between them is exactly this. One puts\n", + "the caret on the token it can see." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-48", + "metadata": {}, + "outputs": [], + "source": [ + "print(cuts[\"report\"].numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-49", + "metadata": {}, + "source": [ + "The other knows which token you left out, and so can offer a hint and move the caret." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-50", + "metadata": {}, + "outputs": [], + "source": [ + "print(cuts[\"require\"].numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-51", + "metadata": {}, + "source": [ + "`maybe_suggest_missing_token_insertion`, then `add_location_if_nearby`, then the error. The\n", + "`name` case went through the first function, because `expected ',' or ';'` names two possible\n", + "tokens and `type_is_unique` is false, and you cannot suggest inserting one of two things. So no\n", + "hint, no swap, and the caret stays on the `b`.\n", + "\n", + "One error message, one extra token in the program, and a whole different code path. That is\n", + "what the column 23 against column 25 split was telling you.\n", + "\n", + "## Three mistakes, two errors\n", + "\n", + "Next irritation. Here is a function with three missing semicolons in it.\n", + "{claim(\"three missing semicolons produce two errors, not three\")}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-52", + "metadata": {}, + "outputs": [], + "source": [ + "one = rec[\"recovery\"]\n", + "\n", + "print(one.source)\n", + "print(one.text)\n", + "print(f\"{len(one.errors)} errors for three mistakes\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-53", + "metadata": {}, + "source": [ + "Two errors for three mistakes, and the second one is not about a semicolon at all. It says\n", + "`expected declaration or statement at end of input`, pointing at the closing brace on line 6,\n", + "which is a fine and correct closing brace.\n", + "\n", + "Two things happened. The first is [error recovery](https://github.com/tamnd/gcc-internals/blob/main/GLOSSARY.md#error-recovery): after complaining on line 4 the\n", + "parser threw tokens away looking for somewhere to start again, and what it found was the end of\n", + "the function. The second is a latch." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-54", + "metadata": {}, + "outputs": [], + "source": [ + "print(cuts[\"latch\"].numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-55", + "metadata": {}, + "source": [ + "`if (parser->error) return false;` and then `parser->error = true;`. Once the parser has\n", + "complained, it will not complain again until something clears the flag, which is what stops one\n", + "missing brace from producing four hundred errors.\n", + "\n", + "The rule to carry away is short. **The first error is the one to trust.** Everything after it\n", + "was produced by a parser that had already lost its place, and the usual outcome of fixing the\n", + "first one is that the rest go away.\n", + "\n", + "## The bracket it points back at\n", + "\n", + "One more piece of the diagnostic structure, because it is the one people notice and cannot\n", + "explain. Here is an unclosed bracket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-56", + "metadata": {}, + "outputs": [], + "source": [ + "one = rec[\"paren\"]\n", + "\n", + "print(one.source)\n", + "print(one.text)" + ] + }, + { + "cell_type": "markdown", + "id": "f03-57", + "metadata": {}, + "source": [ + "Look at what that drew. A caret on line 4, a `~` under the `(` on line 4, and another `~` under\n", + "the `g` on line 5. Three places, one message, two lines apart.\n", + "one diagnostic can point at three places on two lines." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-58", + "metadata": {}, + "outputs": [], + "source": [ + "error = one.errors[0]\n", + "\n", + "print(f\"caret {error.at}\")\n", + "print(f\"fix-it {error.fixes[0].at}, inserting {error.fixes[0].insert!r}\")\n", + "for span in error.related:\n", + " print(f\"secondary {span} {one.line(span.line)!r}\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-59", + "metadata": {}, + "source": [ + "That is `matching_location` in `c_parser_require`, the argument the last section walked past.\n", + "When a parser function asks for a closing bracket it passes the location of the opening one, and\n", + "`add_location_if_nearby` folds it into the same diagnostic if it will fit on the display.\n", + "\n", + "Which is why the message is `expected ')' before 'g'` and not `expected ')'`: three locations,\n", + "one sentence, and enough for you to see the bracket you left open without going to look for it.\n", + "\n", + "## The deepest it ever looks\n", + "\n", + "The parser has four slots. The obvious question is what needs the fourth, and the answer is not\n", + "a C construct." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-60", + "metadata": {}, + "outputs": [], + "source": [ + "one = rec[\"conflict\"]\n", + "\n", + "print(one.source)\n", + "print(one.text)" + ] + }, + { + "cell_type": "markdown", + "id": "f03-61", + "metadata": {}, + "source": [ + "Three errors, all the same sentence, each one seven columns wide. That is not a parse error. It\n", + "is GCC recognising that somebody committed a merge conflict." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-62", + "metadata": {}, + "outputs": [], + "source": [ + "print(cuts[\"conflict\"].numbered())" + ] + }, + { + "cell_type": "markdown", + "id": "f03-63", + "metadata": {}, + "source": [ + "`c_parser_peek_2nd_token`, then `peek_nth_token (parser, 3)`, then `peek_nth_token (parser, 4)`.\n", + "A run of seven `<` characters arrives from the lexer as three `CPP_LSHIFT` tokens and one\n", + "`CPP_LESS`, so recognising it takes four tokens, which is exactly how many there are. Then a\n", + "column check, because a conflict marker is at the start of a line and `a << b << c << d` is not.\n", + "\n", + "Count how often the parser looks that far." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-64", + "metadata": {}, + "outputs": [], + "source": [ + "look = rec.lookahead\n", + "\n", + "print(f\"{look.slots} token slots in the struct\")\n", + "print(f\"{look.peeks} calls to c_parser_peek_token, which looks at one\")\n", + "print(f\"{look.seconds} calls to c_parser_peek_2nd_token, which looks at two\")\n", + "print()\n", + "for depth in sorted(look.depths):\n", + " print(f\"{look.depths[depth]:>3} calls to peek_nth_token with a constant {depth}\")\n", + "print(f\"deepest constant peek: {look.deepest}\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-65", + "metadata": {}, + "source": [ + "Seven hundred and twenty nine one token peeks, ninety nine two token peeks, and eleven that name\n", + "a depth as a constant. Three of those eleven are the four deep ones, and all three are in the\n", + "conflict marker function you have on the screen.\n", + "\n", + "So the fourth slot exists for a thing that is not part of C, and the entire C language is parsed\n", + "in three. That is not a criticism of anybody. It is what a language designed to be compiled in\n", + "one pass on a PDP-11 looks like from the inside, and it is the reason the error messages are the\n", + "shape they are: a parser that can see three tokens cannot know what you meant.\n", + "\n", + "## The size of the thing\n", + "\n", + "Two counts to finish, both taken off the pinned tree rather than described." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-66", + "metadata": {}, + "outputs": [], + "source": [ + "grammar = rec.grammar\n", + "parts = grammar.dialects\n", + "\n", + "print(f\"{len(grammar)} functions named c_parser_* in gcc/c/c-parser.cc\")\n", + "print()\n", + "for name, names in sorted(parts.items(), key=lambda kv: -len(kv[1])):\n", + " share = 100 * len(names) / len(grammar)\n", + " print(f\"{name:<22}{len(names):>4} {share:4.1f}%\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-67", + "metadata": {}, + "source": [ + "Under half the C parser is for C. There are more functions for OpenMP than for the language, and\n", + "the total is three hundred once you add OpenACC, Objective-C and transactional memory. If you go\n", + "looking for the function that parses a `while` loop and find yourself in a file whose contents\n", + "are mostly pragma directives for a parallelism standard, that is why.\n", + "\n", + "Here is a sample of the ones that are about C, so the naming convention is visible." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-68", + "metadata": {}, + "outputs": [], + "source": [ + "wanted = (\"declaration\", \"statement\", \"expression\", \"initializer\", \"struct\", \"label\", \"typeof\")\n", + "\n", + "for name in grammar.named(*wanted):\n", + " print(f\" {name}\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-69", + "metadata": {}, + "source": [ + "`c_parser_` and then the name of the thing in the grammar. Once you know that, finding the code\n", + "for any construct in C is one search, which is the practical thing this section is for.\n", + "\n", + "## Same parser, different machine\n", + "\n", + "One last check, and it is the one that says which of the facts above are about C and which are\n", + "about this laptop. Three of the programs were also compiled by an x86-64 Linux GCC of the same\n", + "release, through Compiler Explorer.\n", + "the same program produces character for character the same diagnostics on two targets." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f03-70", + "metadata": {}, + "outputs": [], + "source": [ + "shared = [one for one in rec if one.elsewhere]\n", + "\n", + "for one in shared:\n", + " print(f\"{one.name:<20}{'same' if one.agrees else 'DIFFERENT'} {one.about}\")\n", + "\n", + "print()\n", + "print(f\"{len(shared)} programs compiled twice, on {rec.target} and on x86-64 Linux\")" + ] + }, + { + "cell_type": "markdown", + "id": "f03-71", + "metadata": {}, + "source": [ + "Identical. Which is the shape of the subject, the same as it was in F02: whether `A * b;` is a\n", + "declaration is a fact about C, and it will read the same on a compiler for a processor nobody\n", + "has built yet. The target gets a say in what `int` is worth. It gets no say at all in what the\n", + "parser thinks you wrote.\n", + "\n", + "## Boss fight\n", + "\n", + "No new tools. Three questions, answerable from the recording:\n", + "\n", + "1. Eight programs, one missing semicolon each, eight different sentences. Seven of them put the\n", + " caret at column 23. Which one does not, and what is different about it?\n", + "2. Thirteen phrases can end a parser complaint, and two of the eight recorded messages cannot be\n", + " pinned to one of them from the text alone. Which two, and why not?\n", + "3. The parser has four token slots. How many calls in the whole file pass a constant 4 to\n", + " `c_parser_peek_nth_token`, and what are they all for?\n", + "\n", + "Then check yourself:\n", + "\n", + "```text\n", + "python lessons/f03-four-tokens/grade.py\n", + "```\n", + "\n", + "or, from a checkout, `just grade f03-four-tokens`. It takes answers on the command line too, so\n", + "`--odd name --unsure char,name --deep 3` is a whole submission. Every answer it marks against is\n", + "worked out from the recording rather than written down.\n", + "\n", + "## What to read next\n", + "\n", + "F04 is GENERIC, which is what this parser produces. The parser has been throwing tokens away\n", + "this whole lesson and building something out of them, and F04 is the something.\n", + "\n", + "F02 is the lesson before this one, if you have not done it. It is where the tokens come from.\n", + "\n", + "If you have a GCC on your machine, every measurement here is one command.\n", + "`printf 'int f(void) { return 1 }' | gcc -fsyntax-only -xc -` is the first one, and\n", + "`-fdiagnostics-format=sarif-stderr` on the end of it is the structure behind the sentence. Change\n", + "the `}` to a `1` and watch the column move." + ] + } + ], + "metadata": { + "colab": { + "provenance": [] + }, + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "name": "python", + "version": "3.12" + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} diff --git a/lessons/f03-four-tokens/grade.py b/lessons/f03-four-tokens/grade.py new file mode 100644 index 0000000..dff4ca8 --- /dev/null +++ b/lessons/f03-four-tokens/grade.py @@ -0,0 +1,154 @@ +"""The F03 boss fight, graded. + +Three questions about the recorded parser diagnostics. Work them out on paper first, then run +this. It says which ones you got right and shows the evidence for each, so a wrong answer is +something to go and look at rather than a mark. + + python lessons/f03-four-tokens/grade.py + python lessons/f03-four-tokens/grade.py --odd name --unsure char,name --deep 3 + +Nothing here is hardcoded. Every answer is worked out from `corpora/diag/f03.json`, so +re-recording against a newer compiler cannot leave the grader marking against a stale answer +key. That has to be true or the grader is worse than no grader. +""" + +from __future__ import annotations + +import argparse +import sys +from collections import Counter + +from gxray import cparse + +#: The eight programs that all leave out the same semicolon. Named rather than derived, +#: because what makes them one family is the mistake they share and no property of the +#: recording says so. +SAME = ("brace", "name", "number", "string", "char", "keyword", "pragma", "eof") + + +def questions() -> dict: + """The three answers, read off the recording rather than written down.""" + rec = cparse.load("f03") + errors = {name: rec[name].errors[0] for name in SAME} + #: The column seven of the eight agree on, found by counting rather than by naming it, + #: so a recompile that moves every caret two to the right still grades correctly. + columns = Counter(one.at.column for one in errors.values()) + usual, _ = columns.most_common(1)[0] + odd = [name for name, one in errors.items() if one.at.column != usual] + unsure = sorted(name for name, one in errors.items() if len(one.suffixes) != 1) + look = rec.lookahead + return { + "rec": rec, + "errors": errors, + "usual": usual, + "odd": odd[0] if len(odd) == 1 else "", + "unsure": unsure, + "deep": look.depths.get(look.deepest, 0), + "look": look, + } + + +def ask(question: str, given: str | None) -> str: + """Take the answer from the command line, or ask for it if it was not given.""" + if given is not None: + return given.strip() + if not sys.stdin.isatty(): + return "" + return input(f"{question} ").strip() + + +def words(text: str) -> list[str]: + """`a, b c` as three names, in any order and with any punctuation between them.""" + return sorted(part.strip() for part in text.replace(",", " ").split() if part.strip()) + + +def mark(label: str, correct: bool, expected: str, evidence: list[str]) -> bool: + print(f"\n{'right' if correct else 'wrong'} {label}") + if not correct: + print(f" the answer is {expected}") + for line in evidence: + print(f" {line}") + return correct + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--odd", help="the one program whose caret is somewhere else") + parser.add_argument("--unsure", help="the messages that do not say which branch made them") + parser.add_argument("--deep", help="how many calls pass a constant 4 to peek_nth_token") + args = parser.parse_args(argv) + + key = questions() + errors, look = key["errors"], key["look"] + + print(f" {len(SAME)} programs, one missing semicolon each") + print(f" {len({one.message for one in errors.values()})} different sentences out of them") + print(f" {len(cparse.SUFFIXES)} phrases c_parse_error can finish a complaint with") + print(f" {look.slots} token slots in the parser") + + where = f"\nSeven carets are at column {key['usual']}. Which program's is not?" + said_odd = ask(where, args.odd) + said_unsure = ask("Which messages do not say which branch made them?", args.unsure) + said_deep = ask("How many calls pass a constant 4 to c_parser_peek_nth_token?", args.deep) + + other = errors[key["odd"]] if key["odd"] else None + example = next(one for name, one in errors.items() if name != key["odd"]) + scored = [ + mark( + "the caret that went somewhere else", + said_odd.strip() == key["odd"], + key["odd"], + [ + f"{example.message} has its caret at column {example.at.column}" + f" and a fix-it inserting {example.fixes[0].insert!r}", + f"{other.message} has its caret at column {other.at.column} and no fix-it", + "the message names two possible tokens, so type_is_unique is false and" + " maybe_suggest_missing_token_insertion is never called", + "no hint means no swap, so the caret stays on the token that upset the parser" + " rather than moving to where the repair would go", + ], + ), + mark( + "the messages that do not say", + words(said_unsure) == key["unsure"], + ", ".join(key["unsure"]), + [f"{name}: {errors[name].message}" for name in key["unsure"]] + + [ + "both end in one character inside single quotes", + "one is a character constant and the other is a one letter identifier" + " quoted back by %qE, and the two branches print the same thing", + "GCC had the token and knew. A recording of the sentence does not", + ], + ), + mark( + "how deep the parser ever looks", + said_deep.strip() == str(key["deep"]), + str(key["deep"]), + [ + f"{look.peeks} peeks at one token, {look.seconds} at two", + ", ".join(f"{n} at a constant {d}" for d, n in sorted(look.depths.items())), + f"all {key['deep']} of the deepest are in c_parser_peek_conflict_marker", + "seven '<' characters lex as three CPP_LSHIFT and one CPP_LESS, which is four" + " tokens, which is the whole buffer", + "so the fourth slot exists for something that is not C, and C is parsed in three", + ], + ), + ] + + got = sum(scored) + print(f"\n{got} of 3.") + if got == 3: + print("The one worth sitting with is the first. Two programs, one mistake, and the") + print("caret lands in two different places because of whether GCC could name the") + print("token you left out. The message is not a description of the error. It is a") + print("readout of how much the parser knew at the moment it gave up.") + return 0 + print("\nThe whole recording is three lines away if you want to look at it:") + print(" from gxray import cparse") + print(' rec = cparse.load("f03")') + print(' print(rec["brace"].text)') + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/lessons/f03-four-tokens/record.py b/lessons/f03-four-tokens/record.py new file mode 100644 index 0000000..b9dd459 --- /dev/null +++ b/lessons/f03-four-tokens/record.py @@ -0,0 +1,675 @@ +"""Record what GCC's C parser says about fifteen small programs, and the code that says it. + + python lessons/f03-four-tokens/record.py + python lessons/f03-four-tokens/record.py --check + +Three kinds of evidence go into `corpora/diag/f03.json`: + + cases fifteen programs through `-fsyntax-only`, each one recorded twice, as the text + a person reads and as the SARIF a machine can assert against. Three of them go + through an x86-64 Linux compiler as well, because a parser is the one part of + GCC that has no target and the cheapest way to say so is to show it + grammar every `c_parser_*` function defined in `gcc/c/c-parser.cc`, counted out of the + pinned tree, which is how the lesson can say what fraction of the C parser is + about C + lookahead every call to the three peeking helpers, and the constant each deep one passes, + which is the only honest way to answer how far ahead the parser can see + +Nothing here links, assembles or optimizes. `-fsyntax-only` stops after the front end, which +is the entire subject, and it is also what makes it safe to record a file with a version +control conflict marker in it. + +The x86-64 half goes through the Compiler Explorer API and is cached in +`tools/cecache/store`, so this is a live request once and a file lookup afterwards. + +`--check` re-asserts every fact the notebook states, against what is already committed and +without running a compiler, which is what the test suite calls. +""" + +from __future__ import annotations + +import argparse +import json +import os +import re +import shutil +import subprocess +import sys +import tempfile +from datetime import date +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT)) + +from gxray import cparse, source # noqa: E402 +from gxray.driver import CEBackend # noqa: E402 +from tools.cecache import Cache # noqa: E402 +from tools.refcheck import GCC_ROOT, PINNED_TAG # noqa: E402 + +OUT = ROOT / "corpora" / "diag" / "f03.json" +CUTS = ROOT / "corpora" / "source" / "f03.json" + +#: The local compiler, the same Homebrew GCC 16.2.0 that F01 and F02 recorded. +GCC = os.environ.get("GXRAY_GCC", "gcc-16") + +#: The other one. Same release, x86-64 Linux, reached through Compiler Explorer. +CE_COMPILER = "cg162" + +#: The one filename every recorded program is compiled under, so that a message from here +#: and a message from a machine on another continent differ in what they say and not in what +#: the file was called on the day. +NAME = "p.c" + +#: The eight programs that make the same mistake. Every one of them leaves out one semicolon +#: after `return 1`, and every one of them gets a different sentence, because the sentence is +#: finished by `c_parse_error` according to the type of the token the parser is looking at. +#: Seven put the caret in the same place and one does not, which is the section. +SAME_MISTAKE: tuple[tuple[str, str, str], ...] = ( + ("brace", "the next token is a close brace", "int f(void) { return 1 }\n"), + ("name", "the next token is an identifier", "int f(void) { int a = 1 b; }\n"), + ("number", "the next token is a number", "int f(void) { return 1 2; }\n"), + ("string", "the next token is a string", 'int f(void) { return 1 "x"; }\n'), + ("char", "the next token is a character constant", "int f(void) { return 1 'c'; }\n"), + ("keyword", "the next token is a keyword", "int f(void) { return 1 while (0); }\n"), + ( + "pragma", + "the next token is a pragma, which is one token", + "int f(void) { return 1\n#pragma GCC unroll 2\n; }\n", + ), + ("eof", "there is no next token", "int f(void) { return 1\n"), +) + +#: The rest of the programs. Each is a whole file, the flags it needs, and whether the same +#: file is worth sending to the other target. +PROGRAMS: tuple[tuple[str, str, str, tuple[str, ...], bool], ...] = ( + ( + "meaning-typedef", + "A is a type, so line 3 declares b", + "typedef int A;\nint b;\nvoid f(void) { A * b; }\n", + ("-Wall", "-Wshadow"), + True, + ), + ( + "meaning-variable", + "A is a variable, so the same line 3 multiplies", + "int A;\nint b;\nvoid f(void) { A * b; }\n", + ("-Wall", "-Wshadow"), + True, + ), + ( + "scope-typedef", + "T is a type at file scope and a variable inside a for header that has closed", + "typedef int T;\nvoid f(void)\n{\n for (int T;;)\n if (1)\n ;\n T *x;\n}\n", + ("-Wall",), + False, + ), + ( + "scope-variable", + "The same program with the two declarations of T swapped over", + "int T;\nvoid f(void)\n{\n for (typedef int T;;)\n if (1)\n ;\n T *x;\n}\n", + ("-Wall",), + False, + ), + ( + "recovery", + "Three missing semicolons, and not three errors", + "void f(void)\n{\n int a = 1\n int b = 2\n int c = 3\n}\n", + (), + False, + ), + ( + "paren", + "A bracket that never closes, and the second place the message points", + "void g(void);\nvoid f(int x)\n{\n if (x\n g();\n}\n", + (), + False, + ), + ( + "conflict", + "The deepest the C parser ever looks ahead, and what it looks for", + "int f(void)\n{\n<<<<<<< HEAD\n return 1;\n=======\n return 2;\n>>>>>>> other\n}\n", + (), + False, + ), +) + +#: Where each span comes from. `first` and `last` are inclusive lines in the pinned tree and +#: `cite` is asserted against `first`, so a span that moves cannot keep a citation pointing +#: at code that is no longer there. +SPANS: tuple[dict, ...] = ( + { + "name": "intermediates", + "path": "gcc/c/c-parser.h", + "first": 26, + "last": 35, + "about": "What sits between libcpp and the parser, and the wish at the end of it", + "cite": "gcc/c/c-parser.h:26@releases/gcc-16.2.0", + }, + { + "name": "slots", + "path": "gcc/c/c-parser.cc", + "first": 188, + "last": 198, + "about": "The parser's entire memory of your file, and the comment that undercounts it", + "cite": "gcc/c/c-parser.cc:188@releases/gcc-16.2.0", + }, + { + "name": "handover", + "path": "gcc/c/c-parser.cc", + "first": 332, + "last": 349, + "about": "The one call. Everything the parser will ever know arrives through it", + "cite": "gcc/c/c-parser.cc:332@releases/gcc-16.2.0", + }, + { + "name": "lookup", + "path": "gcc/c/c-parser.cc", + "first": 467, + "last": 491, + "about": "The symbol table decides what an identifier is, while it is being lexed", + "cite": "gcc/c/c-parser.cc:467@releases/gcc-16.2.0", + }, + { + "name": "conflict", + "path": "gcc/c/c-parser.cc", + "first": 1016, + "last": 1054, + "about": "The only thing in C that needs a fourth token of lookahead", + "cite": "gcc/c/c-parser.cc:1016@releases/gcc-16.2.0", + }, + { + "name": "position", + "path": "gcc/c/c-parser.cc", + "first": 1006, + "last": 1014, + "about": "The guard that decides where an at-end-of-input message puts its caret", + "cite": "gcc/c/c-parser.cc:1006@releases/gcc-16.2.0", + }, + { + "name": "latch", + "path": "gcc/c/c-parser.cc", + "first": 1072, + "last": 1094, + "about": "Two lines, and the reason three mistakes are not three errors", + "cite": "gcc/c/c-parser.cc:1072@releases/gcc-16.2.0", + }, + { + "name": "report", + "path": "gcc/c/c-parser.cc", + "first": 1130, + "last": 1141, + "about": "The other way to report, which puts the caret on the token it can see", + "cite": "gcc/c/c-parser.cc:1130@releases/gcc-16.2.0", + }, + { + "name": "require", + "path": "gcc/c/c-parser.cc", + "first": 1278, + "last": 1318, + "about": "Where the caret goes when GCC knows which token you left out", + "cite": "gcc/c/c-parser.cc:1278@releases/gcc-16.2.0", + }, + { + "name": "loop", + "path": "gcc/c/c-parser.cc", + "first": 2065, + "last": 2099, + "about": "The whole of the C parser's top level, which is six lines", + "cite": "gcc/c/c-parser.cc:2065@releases/gcc-16.2.0", + }, + { + "name": "reclassify", + "path": "gcc/c/c-parser.cc", + "first": 2321, + "last": 2341, + "about": "The patch-up a bug report bought, and the scope it puts right", + "cite": "gcc/c/c-parser.cc:2321@releases/gcc-16.2.0", + }, + { + "name": "suffixes", + "path": "gcc/c-family/c-common.cc", + "first": 7000, + "last": 7013, + "about": "Where a parser's complaint becomes a sentence, and the first branch of it", + "cite": "gcc/c-family/c-common.cc:7000@releases/gcc-16.2.0", + }, + { + "name": "insertion", + "path": "gcc/c-family/c-common.cc", + "first": 9973, + "last": 9997, + "about": "The seven tokens GCC will offer to write for you, and which side they go", + "cite": "gcc/c-family/c-common.cc:9973@releases/gcc-16.2.0", + }, + { + "name": "swap", + "path": "gcc/c-family/c-common.cc", + "first": 9999, + "last": 10046, + "about": "GCC explaining, in a comment with a diagram, why the caret is on the line above", + "cite": "gcc/c-family/c-common.cc:9999@releases/gcc-16.2.0", + }, +) + +#: The GCC test cases the two scope programs are cut down from, named so that a reader who +#: wants the other four variants knows where they are. +SCOPE_TESTS = ("gcc/testsuite/gcc.dg/pr67784-1.c", "gcc/testsuite/gcc.dg/pr67784-2.c") + + +class Watched(Cache): + """The ordinary cache, keeping a note of which entries went through it. + + `tools.tier0.orphans` insists the registry accounts for the store exactly, and it works + that out from an experiment's corpus entry. This lesson's requests are not corpus + entries, so the list has to come from here instead. + """ + + def __post_init__(self) -> None: + super().__post_init__() + self.used: list[str] = [] + + def fetch(self, key: str, send) -> dict: + self.used.append(key) + return super().fetch(key, send) + + +def run(argv: list[str], cwd: Path) -> subprocess.CompletedProcess: + return subprocess.run(argv, cwd=cwd, capture_output=True, text=True, timeout=120, check=False) + + +def rename(text: str, was: str) -> str: + """Call every file `p.c`, wherever it was compiled. + + Compiler Explorer names the file it was handed after its own conventions, and a local + run names it after a temporary directory. Neither is a fact about parsing, and leaving + them in would make two recordings of one program look like a disagreement. + """ + return re.sub(rf"^{re.escape(was)}(?=[:\s])", NAME, text, flags=re.MULTILINE) + + +def compile_here(work: Path, text: str, *args: str) -> tuple[str, str]: + """One program through the front end, kept twice. + + The plain run and the SARIF run are two invocations of the same compiler on the same + file with one flag different, which is the only way to have the sentence a person reads + and the structure a test can assert against without one of them being reconstructed. + """ + (work / NAME).write_text(text, encoding="utf-8") + plain = run([GCC, "-fsyntax-only", *args, NAME], work) + machine = run([GCC, "-fsyntax-only", "-fdiagnostics-format=sarif-stderr", *args, NAME], work) + if not machine.stderr.strip(): + raise SystemExit(f"{GCC} printed no SARIF for:\n{text}") + return plain.stderr, machine.stderr + + +def elsewhere(back: CEBackend, text: str, *args: str) -> str: + """The same program through the x86-64 Linux compiler, made comparable with the local one. + + Two things have to come off. Compiler Explorer names the file ``, and it runs + the compiler attached to something it believes is a terminal, so every diagnostic comes + back wrapped in colour escapes. Asking for no colour is not enough on its own, because + which of the two `-fdiagnostics-color` flags wins depends on the order the service + assembles a command line in, which is not this lesson's business to depend on. + """ + result = back.compile(text, "-fsyntax-only", "-fdiagnostics-color=never", *args) + return rename(cparse.plain(result.stderr), "") + + +def grammar() -> dict: + """Every `c_parser_*` function defined in the pinned tree, by name. + + Definitions and not declarations, which is what anchoring the pattern to the start of a + line buys: GCC's C sources put a function's return type on its own line, so a name in + column one is a definition and a name anywhere else is a call or a prototype. + """ + text = (GCC_ROOT / "gcc" / "c" / "c-parser.cc").read_text(encoding="utf-8", errors="replace") + found = re.findall(r"^(c_parser_\w+) \(", text, flags=re.MULTILINE) + return {"functions": sorted(set(found))} + + +def lookahead() -> dict: + """How far past the current token the parser is able to look, counted rather than recalled. + + `depths` counts only the calls that pass a constant. The handful that pass a variable are + left out on purpose and the notebook says why: they are reached with a small constant from + their callers, or they are parsing OpenMP out of a vector of tokens that was lexed in one + go and is not in the four slots at all. + """ + text = (GCC_ROOT / "gcc" / "c" / "c-parser.cc").read_text(encoding="utf-8", errors="replace") + depths: dict[int, int] = {} + for n in re.findall(r"c_parser_peek_nth_token \(parser, (\d+)\)", text): + depths[int(n)] = depths.get(int(n), 0) + 1 + slots = re.search(r"c_token tokens_buf\[(\d+)\]", text) + if slots is None: + raise SystemExit("no tokens_buf in c-parser.cc, so the lookahead cannot be counted") + return { + "peeks": len(re.findall(r"c_parser_peek_token \(", text)), + "seconds": len(re.findall(r"c_parser_peek_2nd_token \(", text)), + "depths": {str(k): v for k, v in sorted(depths.items())}, + "slots": int(slots.group(1)), + } + + +def one_case(work: Path, back: CEBackend, name: str, about: str, text: str, args, share) -> dict: + plain, machine = compile_here(work, text, *args) + found = cparse.parse_sarif(machine) + entry = { + "about": about, + "flags": list(args), + "source": text, + "text": rename(plain, NAME), + "diagnostics": [cparse.stored(one) for one in found], + } + if share: + entry["elsewhere"] = elsewhere(back, text, *args) + return entry + + +def record() -> dict: + version = run([GCC, "--version"], ROOT).stdout.splitlines()[0] + target = run([GCC, "-dumpmachine"], ROOT).stdout.strip() + watched = Watched() + back = CEBackend(CE_COMPILER, cache=watched) + back.version() + + with tempfile.TemporaryDirectory(prefix="f03-") as tmp: + work = Path(tmp) + cases: dict[str, dict] = {} + for name, about, text in SAME_MISTAKE: + share = name == "brace" + cases[name] = one_case(work, back, name, about, text, (), share) + for name, about, text, args, share in PROGRAMS: + cases[name] = one_case(work, back, name, about, text, args, share) + + return { + "recorded": date.today().isoformat(), + "tag": PINNED_TAG, + "compiler": version, + "target": target, + # Which Compiler Explorer entries this recording stands for, written by the code + # that made the requests rather than kept by hand next to it. + "cache": sorted(set(watched.used)), + "cases": cases, + "grammar": grammar(), + "lookahead": lookahead(), + } + + +def spans() -> dict: + for spec in SPANS: + want = f"{spec['path']}:{spec['first']}@{PINNED_TAG}" + if spec["cite"] != want: + print(f"{spec['name']}: cite says {spec['cite']}, the span starts at {want}") + raise SystemExit(1) + wanted = [{k: v for k, v in spec.items() if k != "cite"} for spec in SPANS] + return source.extract(GCC_ROOT, wanted, PINNED_TAG) + + +def check(rec: cparse.Recording) -> list[str]: + """Every fact the notebook states about the recording, asserted here instead of in prose. + + These are statements about GCC 16.2.0 and not about C. That is what this function is + for: the paragraph next door cannot notice it has gone stale. + """ + wrong: list[str] = [] + + def want(condition: bool, saying: str) -> None: + if not condition: + wrong.append(saying) + + #: The eight programs that make one mistake, which is the spine. + names = [name for name, _, _ in SAME_MISTAKE] + said = [rec.case(name).errors[0].message for name in names if rec.case(name).errors] + want( + len(said) == len(names), + f"{len(said)} of the {len(names)} same-mistake programs produced an error, not all", + ) + want( + len(set(said)) == len(names), + f"the {len(names)} programs produced {len(set(said))} distinct sentences, not" + " one each, and the section is that the sentence is chosen by the next token", + ) + for name in names: + first = rec.case(name).errors[0] + want( + first.suffixes != [], + f"{name} says {first.message!r}, which matches no branch of c_parse_error", + ) + unsure = sorted(name for name in names if rec.case(name).errors[0].suffix is None) + want( + unsure == ["char", "name"], + f"the programs whose message does not say which branch made it are {unsure}, and the" + " section says the two that end in one quoted character, because a one-letter" + " identifier and a character constant print the same", + ) + want( + rec.case("keyword").errors[0].message.endswith("'while'"), + "a keyword is no longer reported the way an identifier is, and the section says the" + " parser hands c_parse_error CPP_NAME when it is looking at a keyword", + ) + want( + rec.case("eof").errors[0].message.endswith(" at end of input"), + "running out of file no longer says so, and the section shows the branch that does", + ) + + #: Seven of the eight put the caret in the same place, because seven go through + #: `c_parser_require` and one goes through `c_parser_error`. + columns = {name: rec.case(name).errors[0].at for name in names} + moved = [name for name in names if columns[name].line == 1 and columns[name].column == 23] + want( + sorted(moved) == sorted(set(names) - {"name"}), + f"the programs whose caret landed just after the 1 are {sorted(moved)}, and the" + " section says it is every one except the two-token complaint", + ) + want( + columns["name"].column == 25, + f"the two-token complaint now points at column {columns['name'].column}, and the" + " section says 25, which is where the offending token is", + ) + hinted = sorted(name for name in names if rec.case(name).errors[0].fixes) + want( + hinted == sorted(set(names) - {"name"}), + f"the programs GCC offered to repair are {hinted}, and the section says every one" + " except the two-token complaint, which is the same one whose caret did not move," + " because moving the caret and offering the repair are the same piece of code", + ) + + #: The brace program in detail, which is the one everybody has seen. + brace = rec.case("brace").errors[0] + want( + brace.message == "expected ';' before '}' token", + f"the brace program now says {brace.message!r}", + ) + want(len(brace.fixes) == 1, f"the brace program carries {len(brace.fixes)} fix-its, not 1") + want( + brace.fixes and brace.fixes[0].insert == ";", + "the fix-it for a missing semicolon no longer inserts a semicolon", + ) + want( + brace.fixes and brace.fixes[0].at.column == brace.at.column, + "the caret and the fix-it are no longer in the same place, and the section says the" + " caret was moved to the fix-it", + ) + want( + brace.related != () and brace.related[0].column == 24, + f"the brace the message names is recorded at {list(brace.related)}, and the section" + " says column 24, which is not where the caret is", + ) + want(brace.moved, "the brace program no longer points at two places, which is the section") + want( + rec.case("brace").agrees, + "the two targets no longer say the same thing about a missing semicolon, which" + " would mean the parser has a target after all", + ) + + #: The pair that differ by one word. + typedef, variable = rec.case("meaning-typedef"), rec.case("meaning-variable") + want( + typedef.line(3) == variable.line(3) == "void f(void) { A * b; }", + "the two meaning programs no longer share line 3, which is the whole comparison", + ) + want( + typedef.source.split("\n")[0] != variable.source.split("\n")[0], + "the two meaning programs no longer differ on line 1", + ) + want( + [one.message for one in typedef.warnings] + == [ + "declaration of 'b' shadows a global declaration", + "unused variable 'b'", + ], + f"the typedef program now says {[one.message for one in typedef.warnings]}, and the" + " section says GCC calls line 3 a declaration", + ) + want( + [one.message for one in variable.warnings] == ["statement with no effect"], + f"the variable program now says {[one.message for one in variable.warnings]}, and" + " the section says GCC calls the same line an expression", + ) + want( + typedef.agrees and variable.agrees, + "the two targets no longer agree about what A * b means, which cannot be right", + ) + + #: The pair GCC needed a bug report to get right. + want( + rec.case("scope-typedef").errors == [], + f"the scope program that should compile now says" + f" {[str(one) for one in rec.case('scope-typedef').errors]}", + ) + want( + "unused variable 'x'" in [one.message for one in rec.case("scope-typedef").warnings], + "the scope program that compiles no longer declares x, and the section says T *x is" + " a declaration there because T is the file scope typedef again by then", + ) + undeclared = rec.case("scope-variable").errors + want( + undeclared != [] and "'x' undeclared" in undeclared[0].message, + f"the scope program that should not compile now says" + f" {[str(one) for one in undeclared]}, and the section says x is undeclared because" + " T *x is a multiplication", + ) + + #: Recovery, which is why you fix the first error and recompile. + recovery = rec.case("recovery") + want( + len(recovery.errors) == 2, + f"three missing semicolons now produce {len(recovery.errors)} errors, and the" + " section says two", + ) + want( + recovery.errors[-1].message.endswith(" at end of input"), + f"the second error is now {recovery.errors[-1].message!r}, and the section says the" + " parser skipped to the end of the block and ran out of file looking for it", + ) + want( + recovery.errors[-1].at.line == 6, + f"the at-end-of-input message is now drawn at line {recovery.errors[-1].at.line}," + " and the section says line 6, because a message about running out of file is put" + " wherever the parser last was and there is no end of file to point at", + ) + + #: The matching bracket, which is the other thing a rich location carries. + paren = rec.case("paren") + want( + paren.errors != [] and paren.errors[0].message == "expected ')' before 'g'", + f"the paren program now says {[str(one) for one in paren.errors]}", + ) + want( + paren.errors and len(paren.errors[0].related) == 2, + f"the missing bracket now points at {len(paren.errors[0].related) if paren.errors else 0}" + " other places, and the section says two: the token it was looking at and the" + " bracket it wanted to match", + ) + want( + paren.errors and sorted(one.line for one in paren.errors[0].related) == [4, 5], + f"the other two places are now on lines" + f" {sorted(one.line for one in paren.errors[0].related) if paren.errors else []}," + " and the section says the open bracket on line 4 and the token on line 5", + ) + + #: The deepest lookahead in the parser, and the only thing that needs it. + conflict = rec.case("conflict") + want( + len(conflict.errors) == 3, + f"the conflict marker program now produces {len(conflict.errors)} errors, not 3", + ) + want( + all(one.message == "version control conflict marker in file" for one in conflict.errors), + f"the conflict program now says {[one.message for one in conflict.errors]}", + ) + want( + all(one.at.column == 1 and one.at.width == 7 for one in conflict.errors), + f"the conflict markers are now {[one.at.width for one in conflict.errors]} columns" + " wide, and the section says seven, starting in column one", + ) + + #: The size of the thing, which is the closing section. + parts = rec.grammar.dialects + want( + len(rec.grammar) == 298, + f"c-parser.cc now defines {len(rec.grammar)} parser functions, not 298", + ) + want( + len(parts["C"]) == 113, + f"{len(parts['C'])} of them are for C, and the section says 113", + ) + want( + len(parts["OpenMP"]) > len(parts["C"]), + f"OpenMP now has {len(parts['OpenMP'])} functions against C's {len(parts['C'])}, and" + " the section says the file has more code for OpenMP than for C", + ) + for name in ("c_parser_translation_unit", "c_parser_peek_conflict_marker", "c_parser_require"): + want(name in rec.grammar.functions, f"{name} is no longer defined in c-parser.cc") + + #: How far it can see. + look = rec.lookahead + want(look.slots == 4, f"the token buffer now has {look.slots} slots, not 4") + want(look.deepest == 4, f"the deepest constant peek is now {look.deepest}, not 4") + want( + look.peeks > 7 * look.seconds, + f"the parser peeks at the next token {look.peeks} times and at the one after it" + f" {look.seconds} times, and the section says the second is rare", + ) + want( + look.depths.get(4, 0) == 3, + f"there are now {look.depths.get(4, 0)} calls that ask for a fourth token, not 3", + ) + return wrong + + +def main(argv: list[str] | None = None) -> int: + ap = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + ap.add_argument("--check", action="store_true", help="assert against what is already there") + args = ap.parse_args(sys.argv[1:] if argv is None else argv) + + if not args.check: + if shutil.which(GCC) is None: + print(f"no {GCC} on PATH. Set GXRAY_GCC, or read B01 and build one.") + return 1 + if not (GCC_ROOT / "gcc" / "c" / "c-parser.cc").is_file(): + print(f"no GCC tree at {GCC_ROOT}. Run `just gcc-src` first.") + return 1 + OUT.parent.mkdir(parents=True, exist_ok=True) + OUT.write_text(json.dumps(record(), indent=1) + "\n", encoding="utf-8") + CUTS.parent.mkdir(parents=True, exist_ok=True) + CUTS.write_text(json.dumps(spans(), indent=1) + "\n", encoding="utf-8") + + rec = cparse.load("f03") + shown = source.load_extract("f03") + wrong = check(rec) + for line in wrong: + print(f" {line}") + verb = "checked" if args.check else "wrote" + said = sum(len(one.diagnostics) for one in rec) + print( + f"{verb} {OUT.relative_to(ROOT)}, {len(rec)} programs, {said} diagnostics, " + f"{len(rec.grammar)} parser functions, recorded {rec.recorded}" + ) + print(f"{verb} {CUTS.relative_to(ROOT)}, {len(shown)} spans, {shown.lines()} lines") + return 1 if wrong else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/site/mkdocs.yml b/site/mkdocs.yml index 9a5e26c..15ae4b3 100644 --- a/site/mkdocs.yml +++ b/site/mkdocs.yml @@ -83,6 +83,7 @@ nav: - BP-PIPELINE, the shape of a compilation: blueprints/BP-PIPELINE.md - BP-DRIVER, the program that runs the other programs: blueprints/BP-DRIVER.md - BP-CPP, the preprocessor: blueprints/BP-CPP.md + - BP-CPARSE, the C parser: blueprints/BP-CPARSE.md - BP-GIMPLE, the GIMPLE statement representation: blueprints/BP-GIMPLE.md - BP-CFG, blocks and edges: blueprints/BP-CFG.md - BP-SSA, one definition per name: blueprints/BP-SSA.md diff --git a/tests/test_cparse.py b/tests/test_cparse.py new file mode 100644 index 0000000..5121474 --- /dev/null +++ b/tests/test_cparse.py @@ -0,0 +1,516 @@ +"""The parser diagnostic reader. + +`gxray.cparse` reads GCC's SARIF diagnostics back into things with names on them, and keeps +two tables transcribed out of the C front end. Most of the tests here are the ordinary kind: +a small input written by hand, an awkward case next to it. + +Three are not ordinary, and they are the ones that matter. +`test_the_insertion_table_matches_the_switch_in_the_tree` and +`test_the_suffix_table_has_a_branch_for_every_one_in_the_tree` compare the two transcribed +tables with the source they were transcribed from, because a table that quietly falls behind +the compiler leaves a lesson saying something false with every other test passing. +`test_every_recorded_message_matches_at_least_one_branch` compares them the other direction, +against a recording, because a table can keep its entries and stop describing the output. +""" + +from __future__ import annotations + +import json +import re +from pathlib import Path + +import pytest +from conftest import grader + +from gxray import cparse + +ROOT = Path(__file__).resolve().parent.parent +TREE = ROOT / "vendor" / "gcc" +COMMON = TREE / "gcc" / "c-family" / "c-common.cc" +PARSER = TREE / "gcc" / "c" / "c-parser.cc" +needs_tree = pytest.mark.skipif(not COMMON.is_file(), reason="vendor/gcc is not checked out") + +LESSON = "f03-four-tokens" + +#: The eight programs that all leave out the same semicolon. +SAME = ("brace", "name", "number", "string", "char", "keyword", "pragma", "eof") + + +def recorded() -> cparse.Recording: + return cparse.load("f03") + + +# --------------------------------------------------------------------------- +# Reading SARIF. + + +def sarif(*results: dict) -> str: + return json.dumps({"runs": [{"results": list(results)}]}) + + +def result(message: str, line: int = 1, column: int = 1, end: int = 0, **rest) -> dict: + region = {"startLine": line, "startColumn": column} + if end: + region["endColumn"] = end + return { + "level": "error", + "message": {"text": message}, + "locations": [{"physicalLocation": {"region": region}}], + **rest, + } + + +def test_a_log_with_no_results_in_it_is_a_clean_compilation(): + assert cparse.parse_sarif(sarif()) == [] + + +def test_text_that_is_not_sarif_is_refused_rather_than_read_as_nothing(): + with pytest.raises(cparse.CParseError, match="not a SARIF log"): + cparse.parse_sarif("error: expected ';' before '}' token") + + +def test_a_log_with_no_runs_is_refused_too(): + with pytest.raises(cparse.CParseError, match="no runs"): + cparse.parse_sarif('{"version": "2.1.0"}') + + +def test_a_result_becomes_a_diagnostic_with_its_place_on_it(): + one = cparse.parse_sarif(sarif(result("expected ';'", line=3, column=11, end=12)))[0] + assert one.error + assert one.message == "expected ';'" + assert (one.at.line, one.at.column, one.at.end) == (3, 11, 12) + assert str(one.at) == "3:11-12" + assert str(one) == "3:11-12: error: expected ';'" + + +def test_a_span_with_no_end_is_one_column_wide(): + assert cparse.Span(1, 5).width == 1 + assert str(cparse.Span(1, 5)) == "1:5" + + +def test_a_span_is_at_least_one_column_wide_however_the_numbers_come_out(): + """SARIF gives `endColumn` one past the last character, and GCC sometimes gives neither.""" + assert cparse.Span(1, 5, 5).width == 1 + assert cparse.Span(1, 5, 12).width == 7 + + +def test_sarif_doubles_braces_and_undouble_puts_them_back(): + assert cparse.undouble("expected ';' before '}}' token") == "expected ';' before '}' token" + assert cparse.undouble("a {{ b }} c") == "a { b } c" + + +def test_undoubling_leaves_a_message_with_no_braces_alone(): + assert cparse.undouble("expected ';' before numeric constant") == ( + "expected ';' before numeric constant" + ) + + +def test_a_recorded_message_arrives_undoubled(): + one = cparse.parse_sarif(sarif(result("expected ';' before '}}' token")))[0] + assert one.message == "expected ';' before '}' token" + + +def test_colour_escapes_come_out_and_nothing_else_does(): + coloured = "\x1b[01m\x1b[Kp.c:1:23:\x1b[m\x1b[K \x1b[01;31m\x1b[Kerror:\x1b[m\x1b[K oops" + assert cparse.plain(coloured) == "p.c:1:23: error: oops" + assert cparse.plain("p.c:1:23: error: oops") == "p.c:1:23: error: oops" + + +def test_a_fix_it_is_read_out_of_the_replacements(): + raw = result( + "expected ';'", + fixes=[ + { + "artifactChanges": [ + { + "replacements": [ + { + "deletedRegion": {"startLine": 1, "startColumn": 23}, + "insertedContent": {"text": ";"}, + } + ] + } + ] + } + ], + ) + one = cparse.parse_sarif(sarif(raw))[0] + assert len(one.fixes) == 1 + assert one.fixes[0].insert == ";" + assert str(one.fixes[0]) == "insert ';' at 1:23" + + +def test_a_related_location_on_another_column_means_the_caret_moved(): + raw = result( + "expected ';'", + column=23, + relatedLocations=[{"physicalLocation": {"region": {"startLine": 1, "startColumn": 24}}}], + ) + one = cparse.parse_sarif(sarif(raw))[0] + assert one.moved + + +def test_a_related_location_in_the_same_place_does_not(): + """`'x' undeclared` carries a secondary at the caret, and that is not a swap.""" + raw = result( + "'x' undeclared", + column=6, + relatedLocations=[{"physicalLocation": {"region": {"startLine": 1, "startColumn": 6}}}], + ) + assert not cparse.parse_sarif(sarif(raw))[0].moved + + +def test_a_diagnostic_with_no_related_locations_did_not_move_either(): + assert not cparse.parse_sarif(sarif(result("oops")))[0].moved + + +def test_the_quoted_line_and_the_function_come_off_the_location(): + raw = result( + "expected ';'", + locations=[ + { + "physicalLocation": { + "region": {"startLine": 1, "startColumn": 23}, + "contextRegion": {"snippet": {"text": "int f(void) { return 1 }\n"}}, + }, + "logicalLocations": [{"fullyQualifiedName": "f"}], + } + ], + ) + one = cparse.parse_sarif(sarif(raw))[0] + assert one.snippet == "int f(void) { return 1 }" + assert one.function == "f" + + +def test_storing_a_diagnostic_and_reading_it_back_gives_the_same_thing(): + """The corpus round trip. If this drifts, a re-recording silently loses a field.""" + for one in recorded(): + for diagnostic in one.diagnostics: + assert cparse._stored(cparse.stored(diagnostic)) == diagnostic + + +def test_a_stored_diagnostic_leaves_out_what_it_does_not_have(): + bare = cparse.Diagnostic(level="error", message="oops", at=cparse.Span(1, 1)) + assert cparse.stored(bare) == {"level": "error", "message": "oops", "at": [1, 1, 0]} + + +# --------------------------------------------------------------------------- +# The two transcribed tables. + + +def test_the_suffix_table_is_thirteen_branches_and_one_catch_all(): + assert len(cparse.SUFFIXES) == 13 + assert sum(1 for one in cparse.SUFFIXES if not one.types) == 1 + assert cparse.SUFFIXES[-1].text == " before %qs token" + + +def test_a_message_that_names_a_number_is_pinned_to_one_branch(): + one = cparse.suffix_for("expected ';' before numeric constant") + assert one is not None + assert one.types == ("CPP_NUMBER",) + + +def test_a_message_that_ends_in_one_quoted_character_is_pinned_to_neither(): + """A one letter identifier and a character constant print the same. This is not a defect.""" + found = cparse.suffixes_for("expected ';' before 'c'") + assert len(found) == 2 + assert {one.types[0] for one in found} == {"CPP_CHAR", "CPP_NAME"} + assert cparse.suffix_for("expected ';' before 'c'") is None + + +def test_a_longer_identifier_is_not_mistaken_for_a_character_constant(): + one = cparse.suffix_for("expected ';' before 'while'") + assert one is not None + assert one.types == ("CPP_NAME",) + + +def test_a_pragma_is_not_read_as_an_identifier_despite_the_quotes(): + one = cparse.suffix_for("expected ';' before '#pragma'") + assert one is not None + assert one.types == ("CPP_PRAGMA",) + + +def test_punctuation_lands_in_the_catch_all(): + one = cparse.suffix_for("expected ';' before '}' token") + assert one is not None + assert one.types == () + + +def test_a_message_that_matches_no_branch_at_all_is_not_forced_into_one(): + assert cparse.suffixes_for("'x' undeclared (first use in this function)") == [] + + +def test_the_insertion_table_is_seven_tokens_split_two_and_five(): + assert len(cparse.INSERTION) == 7 + assert sorted(k for k, v in cparse.INSERTION.items() if v == "before") == ["(", "["] + after = sorted(k for k, v in cparse.INSERTION.items() if v == "after") + assert after == [")", ",", ":", ";", "]"] + + +@needs_tree +def test_the_insertion_table_matches_the_switch_in_the_tree(): + """The transcription against the source, in both directions. + + `get_missing_token_insertion_kind` is a switch over seven token types and a default. If + GCC grows an eighth, or moves one from one arm to the other, the lesson's table has to + move with it, and this is the only thing that would notice. + """ + spelling = { + "CPP_OPEN_SQUARE": "[", + "CPP_OPEN_PAREN": "(", + "CPP_CLOSE_PAREN": ")", + "CPP_CLOSE_SQUARE": "]", + "CPP_SEMICOLON": ";", + "CPP_COMMA": ",", + "CPP_COLON": ":", + } + text = COMMON.read_text(encoding="utf-8") + body = text.partition("get_missing_token_insertion_kind (enum cpp_ttype type)")[2] + body = body.partition("\n}\n")[0] + found = {} + side = "before" + for name in re.findall(r"case (CPP_\w+):|return (MTIK_\w+);", body): + label, action = name + if label: + found[label] = side + elif action == "MTIK_INSERT_BEFORE_NEXT": + side = "after" + assert {spelling[k]: v for k, v in found.items()} == cparse.INSERTION + + +@needs_tree +def test_the_suffix_table_has_a_branch_for_every_one_in_the_tree(): + """Every `catenate_messages` in `c_parse_error`, against the transcribed table. + + Compared on the appended string, which is the part a reader sees. A branch added to the + front end that this table does not know about would leave `suffix_for` returning nothing + for a message the lesson prints. + """ + text = COMMON.read_text(encoding="utf-8") + body = text.partition("c_parse_error (const char *gmsgid, enum cpp_ttype token_type,")[2] + body = body.partition("#undef catenate_messages")[0] + found = re.findall(r'catenate_messages \(gmsgid,\s*"([^"]*)"\)', body) + # One branch prints a hex escape, so the C literal has a doubled backslash in it and the + # transcription has the single backslash a reader would see. + said = {one.replace("\\\\", "\\") for one in found} + assert sorted(said) == sorted({one.text for one in cparse.SUFFIXES}) + + +@needs_tree +def test_the_parser_still_has_exactly_four_token_slots(): + """The number the whole lesson is named after, read off the struct rather than believed.""" + text = PARSER.read_text(encoding="utf-8") + found = re.search(r"c_token tokens_buf\[(\d+)\]", text) + assert found + assert int(found.group(1)) == recorded().lookahead.slots == 4 + + +# --------------------------------------------------------------------------- +# The recording. + + +def test_the_recording_has_the_programs_the_lesson_talks_about(): + rec = recorded() + assert len(rec) == 15 + assert set(SAME) <= set(rec.cases) + assert rec.target + assert rec.compiler.startswith("gcc") + + +def test_asking_for_a_program_that_is_not_there_says_what_is(): + with pytest.raises(cparse.CParseError, match="no case called 'nope'"): + recorded().case("nope") + + +def test_a_recording_iterates_over_its_programs(): + rec = recorded() + assert [one.name for one in rec] == list(rec.cases) + assert rec["brace"] is rec.case("brace") + + +def test_asking_for_a_line_that_is_not_in_the_program_says_how_many_there_are(): + with pytest.raises(cparse.CParseError, match="and line 9 was asked for"): + recorded()["brace"].line(9) + + +def test_one_missing_semicolon_gives_eight_different_sentences(): + """The spine of the lesson. Eight copies of one mistake, eight messages.""" + rec = recorded() + said = {rec[name].errors[0].message for name in SAME} + assert len(said) == len(SAME) + + +def test_every_recorded_message_matches_at_least_one_branch(): + """The tables against the output, rather than against the source they came from.""" + rec = recorded() + for name in SAME: + one = rec[name].errors[0] + assert one.suffixes, one.message + + +def test_exactly_two_of_the_eight_do_not_say_which_branch_made_them(): + rec = recorded() + unsure = sorted(name for name in SAME if cparse.suffix_for(rec[name].errors[0].message) is None) + assert unsure == ["char", "name"] + + +def test_the_one_with_no_fix_it_is_the_one_whose_caret_did_not_move(): + """The split the boss fight asks about, checked as one fact rather than two. + + `expected ',' or ';'` names two possible tokens, so `type_is_unique` is false, so no hint + is offered, so the caret is never swapped onto the hint. Every other program gets all + three, and the recording has to keep showing that or the lesson is wrong. + """ + rec = recorded() + hinted = sorted(name for name in SAME if rec[name].errors[0].fixes) + moved = sorted(name for name in SAME if rec[name].errors[0].moved) + assert hinted == moved == sorted(set(SAME) - {"name"}) + + +def test_the_caret_is_at_the_missing_semicolon_except_in_the_one_case(): + rec = recorded() + columns = {name: rec[name].errors[0].at.column for name in SAME} + assert {name for name, column in columns.items() if column == 23} == set(SAME) - {"name"} + assert columns["name"] == 25 + + +def test_the_swapped_diagnostic_keeps_the_place_the_caret_came_from(): + one = recorded()["brace"].errors[0] + assert one.at.column == 23 + assert one.fixes[0].insert == ";" + assert [span.column for span in one.related] == [24] + assert one.moved + + +def test_the_two_readings_of_one_line_differ_only_in_a_declaration_above_it(): + rec = recorded() + a, b = rec["meaning-typedef"], rec["meaning-variable"] + assert a.line(3) == b.line(3) == "void f(void) { A * b; }" + assert a.line(1) != b.line(1) + assert not a.errors and not b.errors + assert [one.message for one in a.warnings] == [ + "declaration of 'b' shadows a global declaration", + "unused variable 'b'", + ] + assert [one.message for one in b.warnings] == ["statement with no effect"] + + +def test_the_scope_pair_differs_in_whether_the_last_line_compiles(): + rec = recorded() + good, bad = rec["scope-typedef"], rec["scope-variable"] + assert not good.errors + assert "unused variable 'x'" in {one.message for one in good.warnings} + assert len(bad.errors) == 1 + assert bad.errors[0].message.startswith("'x' undeclared") + assert bad.errors[0].at.line == 7 + + +def test_three_missing_semicolons_come_out_as_two_errors(): + one = recorded()["recovery"] + assert one.source.count("\n int") == 3 + assert len(one.errors) == 2 + assert one.errors[-1].message.endswith(" at end of input") + assert one.errors[-1].at.line == 6 + + +def test_an_unclosed_bracket_points_at_two_lines_at_once(): + one = recorded()["paren"] + error = one.errors[0] + assert error.message == "expected ')' before 'g'" + assert len(error.related) == 2 + assert sorted({span.line for span in error.related}) == [4, 5] + assert error.fixes[0].insert == ")" + + +def test_a_conflict_marker_is_three_errors_seven_columns_wide(): + one = recorded()["conflict"] + assert len(one.errors) == 3 + assert {error.message for error in one.errors} == {"version control conflict marker in file"} + assert {error.at.column for error in one.errors} == {1} + assert {error.at.width for error in one.errors} == {7} + + +def test_the_caret_line_is_drawn_from_the_span(): + one = recorded()["conflict"] + drawn = one.under(one.errors[0]) + assert drawn == "^~~~~~~" + + +def test_the_programs_shared_with_another_target_all_agree(): + rec = recorded() + shared = [one for one in rec if one.elsewhere] + assert len(shared) == 3 + assert all(one.agrees for one in shared) + + +def test_a_program_that_was_not_shared_agrees_vacuously(): + """`agrees` has to be true for the twelve that were never sent, or the table lies.""" + assert all(one.agrees for one in recorded() if not one.elsewhere) + + +def test_a_program_whose_other_target_said_something_else_does_not_agree(): + one = cparse.Case(name="x", about="", source="", text="error: a", elsewhere="error: b") + assert not one.agrees + + +def test_trailing_space_is_not_a_disagreement(): + """GCC pads the caret line, and Compiler Explorer does not always send the padding.""" + one = cparse.Case(name="x", about="", source="", text="a \n\nb", elsewhere="a\nb\n") + assert one.agrees + + +# --------------------------------------------------------------------------- +# The counts taken off the tree. + + +def test_the_parser_is_mostly_not_about_c(): + """The number that surprises people, and the reason the file is so large.""" + rec = recorded() + parts = rec.grammar.dialects + assert len(rec.grammar) == 298 + assert sum(len(names) for names in parts.values()) == len(rec.grammar) + assert len(parts["OpenMP"]) > len(parts["C"]) + assert len(parts["C"]) < len(rec.grammar) / 2 + + +def test_asking_the_grammar_for_a_prefix_gives_back_full_names(): + found = recorded().grammar.named("declaration") + assert found == ["c_parser_declaration_or_fndef"] + + +def test_the_deepest_the_parser_looks_is_the_width_of_its_buffer(): + look = recorded().lookahead + assert look.deepest == look.slots == 4 + assert look.depths[4] == 3 + + +def test_most_of_the_peeking_is_one_token_deep(): + look = recorded().lookahead + assert look.peeks > 7 * look.seconds + assert look.seconds > sum(look.depths.values()) + + +def test_a_lookahead_that_was_never_filled_in_still_answers(): + assert cparse.Lookahead().deepest == 2 + + +# --------------------------------------------------------------------------- +# The boss fight. + + +def test_the_grader_answers_its_own_three_questions(): + key = grader(LESSON).questions() + assert key["odd"] == "name" + assert key["unsure"] == ["char", "name"] + assert key["deep"] == 3 + assert key["usual"] == 23 + + +def test_the_grader_accepts_the_right_answers(): + module = grader(LESSON) + assert module.main(["--odd", "name", "--unsure", "char, name", "--deep", "3"]) == 0 + + +def test_the_grader_refuses_the_wrong_ones(): + module = grader(LESSON) + assert module.main(["--odd", "brace", "--unsure", "eof", "--deep", "9"]) == 1 diff --git a/tests/test_tier0.py b/tests/test_tier0.py index 3ef53b4..e24fafb 100644 --- a/tests/test_tier0.py +++ b/tests/test_tier0.py @@ -185,9 +185,13 @@ def test_the_store_holds_nothing_nobody_asks_for(): F02 adds seven. Its recorder sends eight keys, the x86-64 macro table, the five expansion demonstrations, one `-H` trace of `#include `, and the same `cg162` probe, which somebody else had already paid for again. + + F03 adds three. Its recorder sends four keys, the three programs it compiles twice so that + the lesson can claim two targets agree, and the `cg162` probe behind them, which by now is + the fourth lesson to have wanted it and the first to have paid nothing for it. """ assert orphans(REGISTRY) == [] - assert len(set().union(*(keys(x) for x in REGISTRY))) == 50 + assert len(set().union(*(keys(x) for x in REGISTRY))) == 53 def _ce_recording_recipes(): diff --git a/tools/cecache/store/2a/2abc7c5ac5fcd1dfa3a01d67b8464971.json b/tools/cecache/store/2a/2abc7c5ac5fcd1dfa3a01d67b8464971.json new file mode 100644 index 0000000..5bd385b --- /dev/null +++ b/tools/cecache/store/2a/2abc7c5ac5fcd1dfa3a01d67b8464971.json @@ -0,0 +1,61 @@ +{ + "asm": [ + { + "labels": [], + "source": null, + "text": "" + } + ], + "code": 1, + "compilationOptions": [ + "-g", + "-o", + "/app/output.s", + "-masm=intel", + "-fno-verbose-asm", + "-S", + "-fdiagnostics-color=always", + "-fsyntax-only", + "-fdiagnostics-color=never", + "/app/example.c" + ], + "downloads": [], + "execTime": 43, + "filteredCount": 0, + "inputFilename": "example.c", + "instructionSet": "amd64", + "labelDefinitions": {}, + "okToCache": true, + "optOutput": [], + "parsingTime": 0, + "popularArguments": {}, + "processExecutionResultTime": 0.04762199893593788, + "stderr": [ + { + "text": ": In function 'f':" + }, + { + "tag": { + "column": 23, + "file": "example.c", + "line": 1, + "severity": 3, + "text": "error: expected ';' before '}' token" + }, + "text": ":1:23: error: expected ';' before '}' token" + }, + { + "text": " 1 | int f(void) { return 1 }" + }, + { + "text": " | ^~" + }, + { + "text": " | ;" + } + ], + "stdout": [], + "timedOut": false, + "tools": [], + "truncated": false +} diff --git a/tools/cecache/store/54/54c4cc6a52013b9b4c21da0c7e74a9bd.json b/tools/cecache/store/54/54c4cc6a52013b9b4c21da0c7e74a9bd.json new file mode 100644 index 0000000..b17a880 --- /dev/null +++ b/tools/cecache/store/54/54c4cc6a52013b9b4c21da0c7e74a9bd.json @@ -0,0 +1,92 @@ +{ + "asm": [ + { + "labels": [], + "source": null, + "text": "" + } + ], + "code": 0, + "compilationOptions": [ + "-g", + "-o", + "/app/output.s", + "-masm=intel", + "-fno-verbose-asm", + "-S", + "-fdiagnostics-color=always", + "-fsyntax-only", + "-fdiagnostics-color=never", + "-Wall", + "-Wshadow", + "/app/example.c" + ], + "downloads": [], + "execTime": 23, + "filteredCount": 0, + "inputFilename": "example.c", + "instructionSet": "amd64", + "labelDefinitions": {}, + "okToCache": true, + "optOutput": [], + "parsingTime": 0, + "popularArguments": {}, + "processExecutionResultTime": 0.04704199999105185, + "stderr": [ + { + "text": ": In function 'f':" + }, + { + "tag": { + "column": 20, + "file": "example.c", + "line": 3, + "severity": 2, + "text": "warning: declaration of 'b' shadows a global declaration [-Wshadow]" + }, + "text": ":3:20: warning: declaration of 'b' shadows a global declaration [-Wshadow]" + }, + { + "text": " 3 | void f(void) { A * b; }" + }, + { + "text": " | ^" + }, + { + "tag": { + "column": 5, + "file": "example.c", + "line": 2, + "severity": 1, + "text": "note: shadowed declaration is here" + }, + "text": ":2:5: note: shadowed declaration is here" + }, + { + "text": " 2 | int b;" + }, + { + "text": " | ^" + }, + { + "tag": { + "column": 20, + "file": "example.c", + "line": 3, + "severity": 2, + "text": "warning: unused variable 'b' [-Wunused-variable]" + }, + "text": ":3:20: warning: unused variable 'b' [-Wunused-variable]" + }, + { + "text": " 3 | void f(void) { A * b; }" + }, + { + "text": " | ^" + } + ], + "stdout": [], + "timedOut": false, + "tools": [], + "truncated": false +} diff --git a/tools/cecache/store/b6/b671a775fa867d6a251b39a9a885057d.json b/tools/cecache/store/b6/b671a775fa867d6a251b39a9a885057d.json new file mode 100644 index 0000000..3175d57 --- /dev/null +++ b/tools/cecache/store/b6/b671a775fa867d6a251b39a9a885057d.json @@ -0,0 +1,60 @@ +{ + "asm": [ + { + "labels": [], + "source": null, + "text": "" + } + ], + "code": 0, + "compilationOptions": [ + "-g", + "-o", + "/app/output.s", + "-masm=intel", + "-fno-verbose-asm", + "-S", + "-fdiagnostics-color=always", + "-fsyntax-only", + "-fdiagnostics-color=never", + "-Wall", + "-Wshadow", + "/app/example.c" + ], + "downloads": [], + "execTime": 34, + "filteredCount": 0, + "inputFilename": "example.c", + "instructionSet": "amd64", + "labelDefinitions": {}, + "okToCache": true, + "optOutput": [], + "parsingTime": 0, + "popularArguments": {}, + "processExecutionResultTime": 0.06459800153970718, + "stderr": [ + { + "text": ": In function 'f':" + }, + { + "tag": { + "column": 18, + "file": "example.c", + "line": 3, + "severity": 2, + "text": "warning: statement with no effect [-Wunused-value]" + }, + "text": ":3:18: warning: statement with no effect [-Wunused-value]" + }, + { + "text": " 3 | void f(void) { A * b; }" + }, + { + "text": " | ~~^~~" + } + ], + "stdout": [], + "timedOut": false, + "tools": [], + "truncated": false +} diff --git a/tools/tier0/experiments.toml b/tools/tier0/experiments.toml index 6060849..448c18f 100644 --- a/tools/tier0/experiments.toml +++ b/tools/tier0/experiments.toml @@ -438,3 +438,23 @@ expansion demonstrations, and the -H trace of one #include . What keeps honest is that recorder's own check, forty odd assertions that encode what the lesson says \ about the committed recording, run before any prose exists and re-run on every push.\ """ + +[[experiment]] +id = "f03-cparse" +kind = "offline" +question = "Where does the C parser put the caret, and what decides the rest of the sentence?" +lessons = ["f03-four-tokens"] +cache = "corpora/diag/f03.json" +why = """ +Offline because the parser has no dump to compare. It produces GENERIC, which is F04's \ +subject, and the readout this lesson uses is the diagnostic, which the comparators here know \ +nothing about: they count basic blocks and phi nodes in a tree dump, and none of the fifteen \ +programs recorded for F03 gets far enough to have either. The travelling half is the cache \ +listed above, sent again by lessons/f03-four-tokens/record.py: three of the programs go \ +through an x86-64 Linux GCC of the same release, and the lesson's closing claim is that \ +their diagnostics come back character for character the same, which is checked rather than \ +asserted. What keeps the rest honest is that recorder's own check, thirty odd assertions \ +about the committed recording, plus tests/test_cparse.py, which compares the two tables the \ +lesson prints against the switch and the branch list they were transcribed from in the \ +pinned tree.\ +"""