From 1e36e1af49950144857f18e500809419e8cbbe9e Mon Sep 17 00:00:00 2001 From: Yury Date: Sun, 11 Oct 2026 00:05:26 +0300 Subject: [PATCH 1/9] fix: leave synthesized URI parameters out of the outline P6 places a parameter synthesized for an undeclared URI variable at its resource's key, so the authored view counted it as written there and the outline listed a `uriParameters` section nobody wrote. Co-Authored-By: Claude Opus 5.5 --- fastraml/views/authored.py | 9 +++++++-- tests/unit/test_service_queries.py | 16 ++++++++++++++++ 2 files changed, 23 insertions(+), 2 deletions(-) diff --git a/fastraml/views/authored.py b/fastraml/views/authored.py index 5e28fd5..a4a1a10 100644 --- a/fastraml/views/authored.py +++ b/fastraml/views/authored.py @@ -125,10 +125,15 @@ def members[P: Placed](owner: Placed, found: Iterable[P]) -> Iterator[P]: def parameters(owner: Placed, written: Mapping[str, Parameter]) -> Iterator[tuple[str, Parameter]]: """The parameters `owner` wrote. A parameter is a record placed at its - key, in the file its shape was written in. + key, in the file its shape was written in; one synthesized for an + undeclared URI variable is placed at its resource's key and written by no one. """ test = _writer(owner) - return ((name, param) for name, param in written.items() if test(param.base.location, param.key_pos)) + return ( + (name, param) + for name, param in written.items() + if not param.synthesized and test(param.base.location, param.key_pos) + ) def secured_by(owner: EndPoint | Operation) -> Iterator[SecurityScheme]: diff --git a/tests/unit/test_service_queries.py b/tests/unit/test_service_queries.py index 11e0bd8..15cd0a2 100644 --- a/tests/unit/test_service_queries.py +++ b/tests/unit/test_service_queries.py @@ -277,6 +277,22 @@ def test_the_outline_holds_what_each_declaration_writes_and_not_what_it_inherits ]), ] # fmt: skip + def test_an_undeclared_uri_variable_is_not_outlined(self, memory_workspace): + # docs/08 § 6.2: P6 synthesizes a `string` for `{part}` and `{x}`, placed + # at the resource's key, which the author wrote; the parameter they did not. + document = '#%RAML 1.0\ntitle: T\n/items/{id}/{part}:\n uriParameters:\n id: string\n get:\n/a/{x}:\n' + workspace, folder = _buffered(memory_workspace, {'api.raml': document}) + snapshot = workspace.snapshot(f'{folder}/api.raml') + section, resource = SymbolKind.SECTION, SymbolKind.RESOURCE + assert _tree(outline.document_symbols(snapshot, f'{folder}/api.raml')) == [ + ('title', SymbolKind.METADATA, 'T'), + ('/items/{id}/{part}', resource, '', [ + ('uriParameters', section, '', [('id', SymbolKind.PARAMETER, 'string')]), + ('get', SymbolKind.METHOD, ''), + ]), + ('/a/{x}', resource, ''), + ] # fmt: skip + def test_a_redeclared_pattern_is_outlined_where_it_was_written(self, memory_workspace): """docs/07 § 4 puts a redeclared `/b/` at P's place after unwrap; the authored view and the outline follow the author, who wrote `/c/` first. From dd75814d213d1feed2d80453f832f7ce68ff9a95 Mon Sep 17 00:00:00 2001 From: Yury Date: Sun, 11 Oct 2026 00:10:07 +0300 Subject: [PATCH 2/9] fix: place the API description finding at its title, without trees `missing-description` placed the API at its fragment's root node, which only a parse retaining the YAML trees has: the language service keeps none for the default rules, so the finding sat at an unknown position and its suppression directive could not apply. It now reads the `title` in force from the model, the one an Overlay or Extension wrote when it replaced it. The lint command therefore keeps the text, which suppression reads, and the trees only when an enabled rule needs them (`Linter.requires_source`). No benchmark workload runs this path: `unwrap+lint` enables every ruleset, which includes a source-sensitive rule. Co-Authored-By: Claude Opus 5.5 --- docs/18-linting.md | 10 +++-- fastraml/cli/lint.py | 4 +- fastraml/skilldata/lint/SKILL.md | 2 +- fastraml/views/lint/rules/style.py | 12 +++--- tests/unit/test_lint.py | 59 ++++++++++++++++++++++++++++-- 5 files changed, 73 insertions(+), 14 deletions(-) diff --git a/docs/18-linting.md b/docs/18-linting.md index 9b8c333..3013cdd 100644 --- a/docs/18-linting.md +++ b/docs/18-linting.md @@ -226,10 +226,12 @@ fastraml lint [--config FILE] [--severity S] [--ruleset NAME] ``` `--list-rules` and `--explain` require no input document. Other invocations -parse each file with `unwrap=True`, `validate=False`, and `retain_source=True`. -Validation remains the `validate` command's responsibility. A parse failure -produces no findings for that file, reports the parser error, and causes a -nonzero result; remaining files are still attempted. +parse each file with `unwrap=True`, `validate=False` and `retain_text=True`, +which source suppression reads (§ 4); `retain_source=True` only when +`Linter.requires_source` (§ 6). Validation remains the `validate` command's +responsibility. A parse failure produces no findings for that file, reports +the parser error, and causes a nonzero result; remaining files are still +attempted. Formats: diff --git a/fastraml/cli/lint.py b/fastraml/cli/lint.py index 07504d4..57d85f9 100644 --- a/fastraml/cli/lint.py +++ b/fastraml/cli/lint.py @@ -73,7 +73,9 @@ def _lint(args: argparse.Namespace) -> int: # noqa: PLR0911, PLR0912 - command findings = [] failed = False for path in args.files: - raml = parse_or_report(args, path, retain_source=True) + # The text serves suppression directives (docs/18 § 4); the trees only + # a source-sensitive rule. + raml = parse_or_report(args, path, retain_text=True, retain_source=linter.requires_source) if raml is None: failed = True continue diff --git a/fastraml/skilldata/lint/SKILL.md b/fastraml/skilldata/lint/SKILL.md index edb267e..1a8afe3 100644 --- a/fastraml/skilldata/lint/SKILL.md +++ b/fastraml/skilldata/lint/SKILL.md @@ -313,7 +313,7 @@ invalid document to trigger, it is the wrong rule. - **`lint needs an unwrapped model`** — you are calling the Python API directly. Parse with `ParseOptions(unwrap=True)`. - **`enabled lint rules need retained source`** — the same, with a rule enabled - that reads the source text. Also pass `retain_source=True`. + that reads the YAML trees. Also pass `retain_source=True`. - **A plugin's rules never run** — discovery is not activation. Name the plugin under `plugins:` in the config. - **`duplicate rule id`** — two rules claim one name and registration refuses diff --git a/fastraml/views/lint/rules/style.py b/fastraml/views/lint/rules/style.py index 0378d00..2395223 100644 --- a/fastraml/views/lint/rules/style.py +++ b/fastraml/views/lint/rules/style.py @@ -354,12 +354,14 @@ def type_(self, ctx: Context, iri: str, base: BaseShape, shape_kind: str) -> Ite def api(self, ctx: Context, iri: str, api: APIFragment) -> Iterable[Finding]: if api.description is not None: return () - root = ctx.raml.source_node(api.location) - position = UNKNOWN if root is None else root.position + # The API has no key of its own: it is placed at the `title` in force, + # which an Overlay or Extension may have written, and read from the + # model, so no rule in the default set needs the trees (docs/18 § 6). + title = api.title + location = api.location if title is None else title.location + position = UNKNOWN if title is None else title.key_pos return ( - ctx.at( - self.meta, 'entity has no description', location=api.location, position=position, iri=iri, entity='API' - ), + ctx.at(self.meta, 'entity has no description', location=location, position=position, iri=iri, entity='API'), ) def endpoint(self, ctx: Context, iri: str, endpoint: EndPoint) -> Iterable[Finding]: diff --git a/tests/unit/test_lint.py b/tests/unit/test_lint.py index 951e287..eb59bb9 100644 --- a/tests/unit/test_lint.py +++ b/tests/unit/test_lint.py @@ -773,10 +773,39 @@ def api(self, ctx, iri, api): 'unknown-position' ] - def test_api_finding_can_be_suppressed_at_the_root_mapping(self, tmp_path): - source = '#%RAML 1.0\n# fastraml: ignore missing-description\ntitle: t\n' + @pytest.mark.parametrize('retention', ['retain_text', 'retain_source']) + @pytest.mark.parametrize('suppressed', [False, True]) + def test_api_finding_is_placed_and_suppressed_at_its_title_without_trees(self, workspace, retention, suppressed): + directive = '# fastraml: ignore missing-description\n' if suppressed else '' + source = '#%RAML 1.0\nversion: v1\n' + directive + 'title: t\n' + linter = Linter(builtin_registry(), Config(extends=(), rules=(RuleSetting(id='missing-description'),))) + assert not linter.requires_source + raml = workspace.document(source, ParseOptions(unwrap=True, **{retention: True})) + expected = [] if suppressed else [({'entity': 'API'}, raml.location, Position(3, 1, 3, 6))] + assert [(finding.info, finding.location, finding.position) for finding in linter.run(raml)] == expected + + @pytest.mark.parametrize('kind', ['Overlay', 'Extension']) + @pytest.mark.parametrize('replaced', [False, True]) + def test_api_finding_follows_the_title_in_force(self, workspace, kind, replaced): + root = workspace( + { + 'api.raml': '#%RAML 1.0\nversion: v1\ntitle: Original\n', + 'changed.raml': f'#%RAML 1.0 {kind}\nextends: api.raml\nusage: Changed\n' + + ('title: Changed\n' if replaced else ''), + } + ) + raml = workspace.parse(root / 'changed.raml', ParseOptions(unwrap=True, retain_text=True)) config = Config(extends=(), rules=(RuleSetting(id='missing-description'),)) - assert not Linter(builtin_registry(), config).run(parsed(source, tmp_path)) + findings = Linter(builtin_registry(), config).run(raml) + place = ((root / 'changed.raml').as_uri(), 4) if replaced else ((root / 'api.raml').as_uri(), 3) + assert [(finding.location, finding.position.line) for finding in findings] == [place] + + def test_api_finding_without_a_title_has_no_invented_position(self, workspace): + raml = workspace.document('#%RAML 1.0\ntitle: T\n', ParseOptions(unwrap=True, retain_source=True)) + raml.entry_point.title = None + config = Config(extends=(), rules=(RuleSetting(id='missing-description'),)) + findings = Linter(builtin_registry(), config).run(raml) + assert [(finding.location, finding.position.is_known) for finding in findings] == [(raml.location, False)] def test_suppression_uses_an_included_files_location(self, workspace): root = workspace( @@ -1492,6 +1521,30 @@ def test_default_warnings_do_not_fail_the_run(self, disk_workspace, capsys): assert 'deprecated-schemas' in output assert 'WARN 0 errors, 1 warning and 1 info finding.' in output + @pytest.mark.parametrize( + ('rules', 'trees'), [([], False), (['prefer-array-expression'], True), (['prefer-array-expression=off'], False)] + ) + def test_trees_are_kept_only_for_a_source_sensitive_rule(self, disk_workspace, capsys, monkeypatch, rules, trees): + # docs/18 § 5: the text always, for the suppression the directive asks. + from fastraml.cli import lint + + source = '#%RAML 1.0\ntitle: T\n# fastraml: ignore deprecated-schemas\nschemas:\n T: string\n' + root = disk_workspace({'api.raml': source}) + models = [] + original = lint.parse_or_report + + def capture(*args, **kwargs): + models.append(original(*args, **kwargs)) + return models[-1] + + monkeypatch.setattr(lint, 'parse_or_report', capture) + arguments = ['lint', str(root / 'api.raml')] + for rule in rules: + arguments.extend(['--rule', rule]) + assert main(arguments) == EXIT_OK + assert [(raml.retain_text, raml.retain_source) for raml in models] == [(True, trees)] + assert 'deprecated-schemas' not in capsys.readouterr().out + def test_human_color_is_tty_only_and_can_be_disabled(self, disk_workspace, capsys, monkeypatch): root = disk_workspace({'api.raml': '#%RAML 1.0\ntitle: t\nschemas:\n U: string\n'}) monkeypatch.delenv('NO_COLOR', raising=False) From ff3f7eb3df77f9fb249c2214aab7560ef35523c9 Mon Sep 17 00:00:00 2001 From: Yury Date: Sun, 11 Oct 2026 00:11:56 +0300 Subject: [PATCH 3/9] fix: fail `bench ab` when this tree's worker fails A worker failure on either side was reported as "not comparable", so a broken current tree read like a feature the base revision lacks. The base side may still be unavailable, and B then completes its rounds; a failure in B ends the command with status 1 and the worker's output. Co-Authored-By: Claude Opus 5.5 --- bench/ab.py | 25 ++++++++++++------ docs/12-performance.md | 3 ++- tests/bench/test_ab.py | 59 ++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 78 insertions(+), 9 deletions(-) create mode 100644 tests/bench/test_ab.py diff --git a/bench/ab.py b/bench/ab.py index d236d9c..5bb5ecf 100644 --- a/bench/ab.py +++ b/bench/ab.py @@ -104,15 +104,24 @@ def run_ab( # noqa: PLR0913 - the suite's own selectors, plus the revision and entry = write(name, corpus_root, scale) for config in configs: sides: dict[str, list[Measurement]] = {'A': [], 'B': []} - try: - for _ in range(rounds): - sides['A'].append(spawn(name, config, entry, repeat, cwd=tree)) + unavailable: str | None = None + for _ in range(rounds): + if unavailable is None: + try: + sides['A'].append(spawn(name, config, entry, repeat, cwd=tree)) + except RuntimeError as err: + # A corpus for a feature the revision lacks can + # fail there; that side has no number. + unavailable = str(err) + # B still runs: a broken current worker is a failure, + # not missing evidence. + try: sides['B'].append(spawn(name, config, entry, repeat, cwd=_ROOT)) - except RuntimeError: - # A corpus for a feature the revision lacks can fail - # there; that side has no number, which is the result. - failed = 'A' if len(sides['A']) == len(sides['B']) else 'B' - print(f'{name}/{config:<20} fails in {failed}; not comparable') + except RuntimeError as err: + print(f'{name}/{config} FAIL in B (this tree):\n{err}') + return 1 + if unavailable is not None: + print(f'{name}/{config} fails in A; not comparable (B succeeded):\n{unavailable}') continue print(_row(f'{name}/{config}', sides['A'], sides['B'])) finally: diff --git a/docs/12-performance.md b/docs/12-performance.md index 5de4ab0..19c7d4c 100644 --- a/docs/12-performance.md +++ b/docs/12-performance.md @@ -302,7 +302,8 @@ a change to a hot path, or to any code a claim is made about: and this tree over one corpus. A time delta counts only if it is larger than the reported noise. Record the allocation delta whether or not the time moved, because it is nearly deterministic. A field added to every shape - shows there and nowhere else. + shows there and nowhere else. A workload the base revision cannot run is + reported as not comparable; one this tree cannot run fails the command. 3. Where the question is a leaf function's constant factor, add or run a `bench micro` case at sizes that cover the function's range. 4. For a new feature, the base revision has no comparable number. Run diff --git a/tests/bench/test_ab.py b/tests/bench/test_ab.py new file mode 100644 index 0000000..29363a0 --- /dev/null +++ b/tests/bench/test_ab.py @@ -0,0 +1,59 @@ +"""`bench ab`: a revision lacking a feature is not a broken current tree (docs/12 § 5).""" + +from __future__ import annotations + +import pytest + +from bench import ab +from bench.harness import Measurement + + +@pytest.mark.parametrize('fail_a', [None, 1, 2]) +@pytest.mark.parametrize('fail_b', [None, 1, 2]) +def test_a_failing_base_is_not_comparable_and_a_failing_current_tree_fails( + tmp_path, monkeypatch, capsys, fail_a, fail_b +): + base, current = tmp_path / 'base', tmp_path / 'current' + monkeypatch.setattr(ab, '_ROOT', current) + monkeypatch.setattr(ab, '_checkout', lambda ref: base) + monkeypatch.setattr(ab, '_imports_from', lambda tree: tree / 'fastraml' / '__init__.py') + git_calls = [] + + def git(*args): + git_calls.append(args) + return 'base' + + monkeypatch.setattr(ab, '_git', git) + counts = {'A': 0, 'B': 0} + roots = [] + + def write(name, root, scale): + roots.append(root) + return root / 'api.raml' + + def spawn(name, config, entry, repeat, *, cwd): + side = 'A' if cwd == base else 'B' + counts[side] += 1 + if counts[side] == (fail_a if side == 'A' else fail_b): + raise RuntimeError(f'{side} worker stderr') + return Measurement(name, config, seconds=1, allocated_bytes=100, max_rss_bytes=None, retained_bytes=50) + + result = ab.run_ab('base', ['feature'], ['unwrap'], scale=0.5, repeat=2, rounds=3, spawn=spawn, write=write) + output = capsys.readouterr().out + assert result == (0 if fail_b is None else 1) + assert output.count('feature/unwrap') == 1 + assert counts['B'] == (3 if fail_b is None else fail_b) + if fail_b is not None: + assert 'feature/unwrap FAIL in B (this tree)' in output + assert 'B worker stderr' in output + elif fail_a is not None: + assert 'feature/unwrap fails in A; not comparable (B succeeded)' in output + assert 'A worker stderr' in output + assert counts['A'] == fail_a + else: + assert 'noise' in output + assert 'not comparable' not in output + # The base worktree and the corpus are removed however the run ends. + assert git_calls[-1] == ('worktree', 'remove', '--force', str(base)) + assert roots + assert not any(root.exists() for root in roots) From 80a4194f3c8e05717e7eca51a1e6a067e39b61c5 Mon Sep 17 00:00:00 2001 From: Yury Date: Sun, 11 Oct 2026 00:12:39 +0300 Subject: [PATCH 4/9] docs: drop superseded service cost figures MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `docs/12` § 5 said a service workload's first-use composition was avoidable only by retaining the trees, which § 5.2 and the recovery plan no longer hold; § 5.2 already says how to read parse, first-use and warm costs. The `docs/21` § 5 latency paragraph quoted a 2026-09-28 run and a compose cache no longer planned. Dated measurements belong in `docs/reports/`, which `docs/README.md` now describes. Co-Authored-By: Claude Opus 5.5 --- docs/12-performance.md | 12 ------------ docs/21-language-service.md | 6 ------ docs/README.md | 3 +++ 3 files changed, 3 insertions(+), 18 deletions(-) diff --git a/docs/12-performance.md b/docs/12-performance.md index 19c7d4c..82e69a1 100644 --- a/docs/12-performance.md +++ b/docs/12-performance.md @@ -311,18 +311,6 @@ a change to a hot path, or to any code a claim is made about: Include the workload, both deltas, and the noise in the commit message. -A service workload's measured region is a cold snapshot: the parse, plus the -first query of each kind, then reuse of the built indices. Read its delta in -two parts. The parse side is where the base revision built and retained every -file's tree and the comparison does not. The first-use side is where a query -composes on demand the tree the base retained at parse time, one composition -per file per snapshot, about a third of a parse of the file; repeated -queries of the same kind are cached and show no delta. The first-use cost is -avoidable only by retaining the trees, the per-snapshot memory cost -(docs/15 § 2). When a change moves composition from parse time to first use, -name which side the delta is on before attributing it to the change's hot -path (docs/21 § 2, docs/21 § 4). - `bench compare` against the committed baseline is only a coarse check for large regressions: the baseline and the comparison run at different times, and the machine drifts in between. Follow § 5.1 before optimizing and § 5.2 when diff --git a/docs/21-language-service.md b/docs/21-language-service.md index 33a44cf..46c65e5 100644 --- a/docs/21-language-service.md +++ b/docs/21-language-service.md @@ -501,12 +501,6 @@ value, so an integer larger than a double reaches a JavaScript client as written. It is the preview's source in `contrib/fastraml-vscode` (`docs/17` § 4). -**Latency.** On `large`, an edit costs 475 ms and allocates 48.8 MB before -its parser diagnostics, against 366 ms for a plain `unwrap+validate` parse -(`python -m bench run --bench large --config service`). About a quarter of -the parse composes the unchanged libraries: the most a compose cache (G8) -could save. - ## 6. Verification - `test_service_text.py`: conversion in each encoding, both ways. diff --git a/docs/README.md b/docs/README.md index 7cff42b..9f1ec21 100644 --- a/docs/README.md +++ b/docs/README.md @@ -67,6 +67,9 @@ architecture. The implementation and these documents must agree. for explaining past decisions. It is not normative. - `research/` contains open investigations. Nothing there defines parser behavior or may be required by the implementation. +- `reports/YYYY-MM-DD/` contains dated benchmark measurements, audits and other + point-in-time records. Results stay in those reports; the numbered documents + describe current contracts and behavior, not individual runs. Current parser work is complete. Deferred work is listed in the non-normative [status and roadmap](15-implementation-plan.md). Conformance status belongs in From 131b08ed6189953e6157dc0f79b3d353e878018c Mon Sep 17 00:00:00 2001 From: Yury Date: Sun, 11 Oct 2026 00:19:05 +0300 Subject: [PATCH 5/9] perf: share an open buffer's text with its snapshot The overlay loader hands the parse a buffer's encoded bytes, so each snapshot retained its own decoded copy of every open file beside the buffer's. Where the two are equal the snapshot now keeps the buffer's string; a buffer with a leading BOM keeps the parse's BOM-free text. `bench ab HEAD --bench service-session --config unwrap`, 5 rounds: time within noise (107.5 %); peak 35.62 -> 35.31 MB (-0.9 %); retained 27.81 -> 27.50 MB (-1.1 %). Co-Authored-By: Claude Opus 5.5 --- fastraml/service/workspace.py | 6 ++++++ tests/unit/test_service_workspace.py | 13 +++++++++++++ 2 files changed, 19 insertions(+) diff --git a/fastraml/service/workspace.py b/fastraml/service/workspace.py index d4c6d88..87a5f74 100644 --- a/fastraml/service/workspace.py +++ b/fastraml/service/workspace.py @@ -438,6 +438,12 @@ def _parse(self, root: str) -> Snapshot: else RamlError.wrap('load resource', err, root, kind=ErrorKind.READING) ) return Snapshot(root, None, failure, frozenset({root})) + texts = raml.source_texts + for uri, kept in texts.items(): + # The parse decoded its own copy of an open buffer's text; keep the + # buffer's instead of a second one for the snapshot's lifetime. + if (buffer := self.buffers.get(uri)) is not None and buffer.text == kept: + texts[uri] = buffer.text return Snapshot( root, raml, diff --git a/tests/unit/test_service_workspace.py b/tests/unit/test_service_workspace.py index 3fdcef6..8ef1ca9 100644 --- a/tests/unit/test_service_workspace.py +++ b/tests/unit/test_service_workspace.py @@ -233,6 +233,19 @@ def test_a_file_no_root_reads_is_parsed_alone(self, tmp_path): assert snapshot.root == f'{folder}/lib.raml' assert snapshot.error is None + def test_a_snapshot_keeps_an_open_buffers_text_not_a_second_copy(self, memory_workspace): + folder = path_to_file_uri(memory_workspace.root) + workspace = Workspace([folder]) + texts = {'api.raml': self.FILES['api.raml'], 'lib.raml': BOM + LIBRARY} + for name, text in texts.items(): + workspace.open(f'{folder}/{name}', text, 1) + snapshot = workspace.snapshot(f'{folder}/api.raml') + assert snapshot.raml is not None + kept = snapshot.raml.source_texts + assert kept[f'{folder}/api.raml'] is workspace.buffers[f'{folder}/api.raml'].text + # The parser drops the leading BOM, so that text is the parse's own. + assert kept[f'{folder}/lib.raml'] == LIBRARY + def test_an_os_error_is_keyed_by_its_errno(self, tmp_path, monkeypatch): """docs/11 § 6: an `OSError` past the parse reports a key, not the OS text.""" workspace, folder = _workspace(tmp_path, {'api.raml': API}) From 069e033845892e17167f6cc7f09fec03f6841b92 Mon Sep 17 00:00:00 2001 From: Yury Date: Sun, 11 Oct 2026 00:44:23 +0300 Subject: [PATCH 6/9] perf: read hover's source keys along the cursor's path The first hover in a file sorted every key the grammar walk yields and scanned each type value for built-in tokens, an index kept for the snapshot's lifetime (15,203 keys and 2.6 MB on the hover corpus). Hover now reads only the keys enclosing the cursor, `syntax.keys_at`, one per mapping level found by binary search, and the built-in tokens of the one type value the cursor is in. A token's offset is taken as its column where the scalar's one-line span is its text or its text in quotes, instead of splitting the file to compare. Over the TCK, the path reaches every key the walk yields. `bench ab HEAD --bench hover --bench service-session --config unwrap`, 5 rounds: - hover: time within noise (16.2 %); peak unchanged; retained 26.29 -> 23.71 MB (-9.8 %). - service-session: 1260.5 -> 1142.8 ms (-9.3 %, noise 7.4 %); peak unchanged; retained 27.50 -> 24.87 MB (-9.6 %). `bench linearity --bench hover`: time 0.992, peak 0.992, retained 0.998. Co-Authored-By: Claude Opus 5.5 --- docs/21-language-service.md | 19 +++--- fastraml/parser/syntax.py | 54 +++++++++++++++- fastraml/service/hover.py | 102 ++++++++++++------------------- tests/bench/test_corpus.py | 4 +- tests/unit/test_service_hover.py | 42 +++++++++++++ 5 files changed, 147 insertions(+), 74 deletions(-) diff --git a/docs/21-language-service.md b/docs/21-language-service.md index 46c65e5..81091fc 100644 --- a/docs/21-language-service.md +++ b/docs/21-language-service.md @@ -25,7 +25,7 @@ and several views, and only `cli/` imports it (`docs/02` § 2; | `service/text.py` | converting a column between fastRAML and a protocol | | `service/queries.py` | the queries, in fastRAML positions | | `service/outline.py` | the outline, over the authorship view (`docs/16` § 10) | -| `service/hover.py`, `service/hoverdocs.py` | author-facing hover, source-key indices and explanatory prose (§ 4.2) | +| `service/hover.py`, `service/hoverdocs.py` | author-facing hover, source-key lookup and explanatory prose (§ 4.2) | | `service/datahover.py` | typed `DataNode` key and value spans, shared across value-bearing sites (§ 4.2) | | `service/index.py` | lazy declaration, ID and reverse-hierarchy lookups shared by snapshot queries (§ 4) | | `service/lenses.py` | code-lens sites and on-demand effective RAML type rendering (§ 4.3) | @@ -326,14 +326,18 @@ extent is not a fallback for an unknown child. Inline JSON has only the encoded scalar's root span; hover does not invent spans for decoded children. Source-only primitive tokens, including those -in an unapplied template, are read through the type-expression parser and -checked against the source text. No source query binds a reference. Explanatory +in an unapplied template, are read through the type-expression parser. A +token's offset is its column only where the scalar's one-line span is its +text, or its text in quotes; where an escape or a tag shifts the columns, no +token is found. No source query binds a reference. Explanatory prose is a documentation catalogue, not a table that accepts fields or overrides parser diagnostics. -Hover indices and formatted subjects are lazy per snapshot. Source keys are indexed once per queried -file from retained nodes, or a composition of its retained text when source -trees were not retained. These nodes and indices die with the snapshot. +Hover indices and formatted subjects are lazy per snapshot. Source keys are not +indexed: each hover reads the path to its cursor (`syntax.keys_at`), one entry +per mapping level found by binary search, in the file's retained nodes, or a +composition of its retained text when source trees were not retained. Those +nodes die with the snapshot. The typed-data token index is owned by the snapshot, lazy and built once; formatted data targets are cached by hover. The same typed-data targets supply go-to-definition for nested field keys and scalar values. A definition request may populate that @@ -517,7 +521,8 @@ written. It is the preview's source in `contrib/fastraml-vscode` - `test_service_hover.py`: contextual field meanings, full Markdown prose, authored and inherited summaries, aliases, optional versus nullable values, source-only template help, token boundaries, opaque data, educational examples, - and user-defined facet descriptions at declarations and supplied keys. + and user-defined facet descriptions at declarations and supplied keys; over + the TCK, the cursor path reaches every key the grammar walk yields. - `test_service_data_hover.py`: shared nested-field and scalar-value help for custom facets, annotations, examples, defaults and enums; array items, inheritance, patterns, discriminated and ambiguous unions, recursive types, diff --git a/fastraml/parser/syntax.py b/fastraml/parser/syntax.py index 11bf789..0dc73bb 100644 --- a/fastraml/parser/syntax.py +++ b/fastraml/parser/syntax.py @@ -2,12 +2,14 @@ These are mapping positions, not annotation targets or a second resolver. `child_site` is the extension merge's existing grammar (docs/19 § 3.1). -`keys` projects a composed source tree into those positions for an editor -(docs/21 § 4.2); data and application arguments stay opaque. +`keys` projects a composed source tree into those positions for an editor, +and `keys_at` the path to one cursor (docs/21 § 4.2); data and application +arguments stay opaque. """ from __future__ import annotations +from bisect import bisect_right from dataclasses import dataclass from enum import Enum, auto from typing import TYPE_CHECKING, Final @@ -32,6 +34,7 @@ 'fragment_site', 'is_media_type_map', 'keys', + 'keys_at', ] #: Resource methods; optional `?` spellings belong to resource-type templates. @@ -197,6 +200,51 @@ def annotation(self) -> bool: return self.site not in NAME_MAPS and is_annotation_key(self.node.value) +_OPAQUE: Final = frozenset({Site.DATA, Site.APPLICATION, Site.GENERIC}) + + +def keys_at(root: Node, site: Site, line: int, column: int, *, table: str = '') -> list[Key]: + """The keys `keys` yields whose entry encloses `line:column`, outermost first. + + Each level follows the last entry written at or before the position, so + one lookup reads one path rather than the file. The last key holds the + position, or the value it lies in, or neither when it lies past the end. + """ + position = (line, column) + found: list[Key] = [] + node = root + while site not in _OPAQUE: + if node.kind is NodeKind.SEQUENCE: + # Linear: an alias item keeps its anchor's earlier position. + start = (node.line, node.column) + items = [item for item in node.content if start <= (item.line, item.column) <= position] + if not items: + break + node = items[-1] + continue + if node.kind is not NodeKind.MAPPING: + break + if site is Site.BODY and not is_media_type_map(node): + if any('/' in key.value for key, _ in pairs(node)): + break + site = Site.TYPE + content = node.content + # A mapping's own entries are in document order. + index = bisect_right( + range(len(content) // 2), position, key=lambda i: (content[2 * i].line, content[2 * i].column) + ) + if not index: + break + key, value = content[2 * index - 2], content[2 * index - 1] + found.append(Key(key, value, node, site, table)) + if key.position.holds(line, column) or (value.line, value.column) < (key.line, key.column): + break + child = child_site(site, key.value) + table = key.value if child in NAME_MAPS else table + node, site = value, child + return found + + def keys(root: Node, site: Site, *, table: str = '') -> Iterator[Key]: """Source keys, without decoding data or expanding templates. @@ -206,7 +254,7 @@ def keys(root: Node, site: Site, *, table: str = '') -> Iterator[Key]: stack = [(root, site, table)] while stack: node, context, table = stack.pop() - if context in (Site.DATA, Site.APPLICATION, Site.GENERIC): + if context in _OPAQUE: continue if node.kind is NodeKind.SEQUENCE: stack.extend( diff --git a/fastraml/service/hover.py b/fastraml/service/hover.py index d724120..31db2b3 100644 --- a/fastraml/service/hover.py +++ b/fastraml/service/hover.py @@ -1,9 +1,10 @@ """Author-facing hover: source explanations and compact model summaries. The service composes the parser's source grammar, occurrences and authorship -view. It binds no name and runs no pass (docs/21 § 4.2). Indices and source -keys are built once per snapshot, so successive hovers do not rescan the whole -model. Formatted subjects and declaration hints are cached. +view. It binds no name and runs no pass (docs/21 § 4.2). Indices are built +once per snapshot, so successive hovers do not rescan the whole model; source +keys are read along the cursor's path only. Formatted subjects and declaration +hints are cached. """ from __future__ import annotations @@ -19,7 +20,7 @@ from fastraml.parser.fragments import APIFragment, LibraryLink from fastraml.parser.includes import is_json_ref from fastraml.parser.security import SecuritySchemeDefinition -from fastraml.parser.syntax import METHODS, NAME_MAPS, Key, Site, child_site, fragment_site, keys +from fastraml.parser.syntax import METHODS, NAME_MAPS, Key, Site, child_site, fragment_site, keys, keys_at from fastraml.parser.templates import TemplateDefinition from fastraml.service.datahover import DataHover, DataTarget, data_roots from fastraml.service.hoverdocs import BUILTINS, METHOD_DOCS, field_doc @@ -34,7 +35,6 @@ underlying_type, ) from fastraml.service.source import original_tree -from fastraml.service.text import Lines from fastraml.types.base import BaseShape, Parameter, facets_of from fastraml.types.complex_ import ArrayShape, ObjectShape, UnionShape, UnknownShape from fastraml.types.expressions import Array, Optional_, Primitive, Union, parse_expression @@ -117,8 +117,6 @@ class Hover: '_descriptions', '_includes', '_inlay_declarations', - '_key_starts', - '_keys', '_mappings', '_nodes', '_occurrences', @@ -130,7 +128,6 @@ class Hover: '_semantic', '_sites', '_source_backend', - '_source_builtins', '_source_generation', '_sources', '_subjects', @@ -159,12 +156,9 @@ def __init__( # noqa: PLR0913 - the snapshot's borrowed source owner and compos self._by_uri: dict[str, list[_Subject]] = {} self._sites: dict[tuple[str, int, int], _Subject] = {} self._mappings: dict[tuple[str, int, int], _Subject] = {} - self._keys: dict[str, list[Key]] = {} - self._key_starts: dict[str, list[tuple[int, int]]] = {} self._resources: dict[tuple[str, int, int], EndPoint] = {} self._operations: dict[tuple[str, int, int], tuple[str, Operation]] = {} self._responses: dict[tuple[str, int, int], Response] = {} - self._source_builtins: dict[str, Occurrences] = {} self._ambiguous_sites: set[tuple[str, int, int]] = set() self._nodes: dict[str, Node | None] = {} self._contexts: dict[str, set[tuple[Site, str]]] = {} @@ -281,7 +275,8 @@ def _bodies(self, bodies: Iterable[Body], role: str) -> None: def at(self, uri: str, line: int, column: int) -> tuple[str, Position] | None: """Explain the precise token under the cursor, or return no answer.""" - key = self._key_at(uri, line, column) + path = self._keys_at(uri, line, column) + key = path[-1] if path and path[-1].node.position.holds(line, column) else None if key is not None: text = self._key_doc(uri, key) if text is not None: @@ -299,9 +294,9 @@ def at(self, uri: str, line: int, column: int) -> tuple[str, Position] | None: data = self._data.at(uri, line, column) if data: return self._data_docs(data), data[0].span - primitive = self._source_builtins[uri].at(uri, line, column) - if primitive: - return self._builtin(primitive[0].written), primitive[0].span + primitive = self._builtin_at(path[-1], line, column) if path and key is None else None + if primitive is not None: + return self._builtin(primitive[0]), primitive[1] if key is not None: subject = self._sites.get((uri, key.node.line, key.node.column)) text = self._describe(subject) if subject is not None else _unbound_doc(key) @@ -475,59 +470,42 @@ def _discover_contexts(self) -> None: contexts.add(context) pending.append((target, child, child_table)) - def _source_keys(self, uri: str) -> None: + def _keys_at(self, uri: str, line: int, column: int) -> list[Key]: + """The source keys enclosing the cursor, read along its path only.""" if uri not in self._contexts and not self._contexts_ready: self._discover_contexts() contexts = self._contexts.get(uri, {(Site.DATA, '')}) context, table = next(iter(contexts)) if len(contexts) == 1 else (Site.DATA, '') if context in (Site.DATA, Site.APPLICATION, Site.GENERIC): - self._keys[uri] = [] - self._key_starts[uri] = [] - self._source_builtins[uri] = Occurrences((), ()) - return + return [] node = self._node(uri) - found = ( - [] - if node is None - else sorted(keys(node, context, table=table), key=lambda key: (key.node.line, key.node.column)) - ) - self._keys[uri] = found - self._key_starts[uri] = [(key.node.line, key.node.column) for key in found] - lines = Lines(self._raml.source_texts.get(uri, '')) - primitives = [] - cache: ExprCache = {} - for key in found: - value = key.type_value() - if value is None: - continue - if value.tag != TAG_STR: - continue - tokens = [] - if value.value in BUILTINS: - tokens.append((value.value, 0)) - elif _OPERATORS.search(value.value): - with suppress(RamlError): - cached = self._raml.expr_cache.get(value.value) - if cached is not None: - cache[value.value] = cached - tree = parse_expression(value.value, cache) - tokens.extend((token.name, token.col) for token in _primitives(tree)) - for name, offset in tokens: - span = value.position.within(value.value).shifted(offset, len(name)) - if lines.line(span.line)[span.column - 1 : span.end_column - 1] == name: - primitives.append( - Occurrence(uri, span.line, span.column, span.end_column, Role.BUILTIN, Kind.TYPE, None, name) - ) - self._source_builtins[uri] = Occurrences(primitives, ()) - - def _key_at(self, uri: str, line: int, column: int) -> Key | None: - if uri not in self._keys: - self._source_keys(uri) - index = bisect_right(self._key_starts[uri], (line, column)) - if index: - found = self._keys[uri][index - 1] - if found.node.position.holds(line, column): - return found + return [] if node is None else keys_at(node, context, line, column, table=table) + + def _builtin_at(self, key: Key, line: int, column: int) -> tuple[str, Position] | None: + """A built-in type named by the type expression the cursor is in.""" + value = key.type_value() + if value is None or value.tag != TAG_STR or value.line != line or value.end_line != line: + return None + text = value.value + # Token offsets map onto columns only where the span is the text, or + # the text in quotes: an escape or a tag shifts them. + if value.end_column - value.column not in (len(text), len(text) + 2): + return None + tokens: list[tuple[str, int]] = [] + if text in BUILTINS: + tokens.append((text, 0)) + elif _OPERATORS.search(text): + with suppress(RamlError): + cache: ExprCache = {} + cached = self._raml.expr_cache.get(text) + if cached is not None: + cache[text] = cached + tokens.extend((token.name, token.col) for token in _primitives(parse_expression(text, cache))) + start = value.position.within(text) + for name, offset in tokens: + span = start.shifted(offset, len(name)) + if span.column <= column < span.end_column: + return name, span return None def _key_doc(self, uri: str, key: Key) -> str | None: diff --git a/tests/bench/test_corpus.py b/tests/bench/test_corpus.py index 4d09bb0..5d5fbc4 100644 --- a/tests/bench/test_corpus.py +++ b/tests/bench/test_corpus.py @@ -180,7 +180,7 @@ def test_inlays_reaches_inferred_types_data_types_and_inherited_facets(self, tmp original = inlays.inlay_hints def source(*args, **kwargs): - pytest.fail('model-backed inlays must not compose source or populate grammar/builtin indices') + pytest.fail('model-backed inlays must not compose source or read its grammar') def hints(snapshot, uri, span): nonlocal passes @@ -190,7 +190,7 @@ def hints(snapshot, uri, span): return result monkeypatch.setattr(Hover, '_node', source) - monkeypatch.setattr(Hover, '_source_keys', source) + monkeypatch.setattr(Hover, '_keys_at', source) monkeypatch.setattr(inlays, 'inlay_hints', hints) entry = corpus.write_hover(tmp_path, family_count=count) run_one('inlays', 'unwrap', entry, repeat=1) diff --git a/tests/unit/test_service_hover.py b/tests/unit/test_service_hover.py index 809ee12..9ee8e1d 100644 --- a/tests/unit/test_service_hover.py +++ b/tests/unit/test_service_hover.py @@ -282,6 +282,18 @@ def test_a_builtin_explains_allowed_values_and_supported_facets(self, hover): assert '`pattern`' in string assert 'RAML reference' in integer + def test_a_builtin_is_found_in_quotes_but_not_where_an_escape_shifts_its_columns(self, memory_workspace): + # docs/21 § 4.2: a token's offset in the expression is its column only + # where the span is the text, or the text in quotes. + document = '#%RAML 1.0\ntitle: T\ntypes:\n A:\n type: "string | nil"\n B:\n type: "n\\x69l"\n' + workspace, folder = _buffered(memory_workspace, {'api.raml': document}) + snapshot = workspace.snapshot(f'{folder}/api.raml') + assert snapshot.error is None + text, span = queries.hover(snapshot, f'{folder}/api.raml', *_where(document, 'nil')) + assert 'Accepts only null' in text + assert (span.column, span.end_column) == (_where(document, 'nil')[1], _where(document, 'nil')[1] + 3) + assert queries.hover(snapshot, f'{folder}/api.raml', *_where(document, 'x69')) is None + def test_hover_does_not_extend_past_the_target_token(self, memory_workspace): document = '#%RAML 1.0\ntitle: T\ntypes:\n Data:\n properties: {} # comment\n' workspace, folder = _buffered(memory_workspace, {'api.raml': document}) @@ -496,3 +508,33 @@ def test_method_help_explains_http_behavior(self, hover): assert 'Retrieves a representation' in text assert 'without requesting a change to server state' in text assert 'possible responses' in text + + +@pytest.mark.tck +def test_the_cursor_path_finds_every_key_the_whole_file_walk_yields(): + # docs/21 § 4.2: hover reads the path to one cursor, which must reach every + # key the grammar walk projects, in the same position and table. + from fastraml.errors import RamlError + from fastraml.parser.syntax import fragment_site, keys, keys_at + from fastraml.yamlnode import compose + from tests.tck.conftest import tck_root + + root = tck_root() + if root is None: + pytest.skip('no TCK corpus; set FASTRAML_TCK_DIR') + checked, missed = 0, [] + for path in sorted(root.rglob('*.raml')): + text = path.read_text(encoding='utf-8-sig', errors='replace') + header = text.partition('\n')[0].removeprefix('#%RAML 1.0').strip() + try: + node = compose(text, uri=path.as_uri()) + except RamlError: + continue + site = fragment_site(header or 'API') + for key in keys(node, site): + found = keys_at(node, site, key.node.line, key.node.column) + checked += 1 + if not found or (found[-1].node, found[-1].site, found[-1].table) != (key.node, key.site, key.table): + missed.append(f'{path.relative_to(root).as_posix()}:{key.node.line}:{key.node.column}') + assert checked > 10_000 + assert not missed, missed[:20] From 7e1474587cc0ff307a5a83588faee889331d56a7 Mon Sep 17 00:00:00 2001 From: Yury Date: Sun, 11 Oct 2026 00:45:25 +0300 Subject: [PATCH 7/9] docs: record the cursor-local hover lookup The dated report holds the scope, reach and measurements; the recovery plan's execution record and docs/15 now name accurate authored section ranges as the remaining candidate, recorded by the parser. Co-Authored-By: Claude Opus 5.5 --- docs/15-implementation-plan.md | 8 +-- .../2026-10-11/service-cursor-hover.md | 54 +++++++++++++++++++ .../2026-10-10/service-recovery-plan.md | 9 +++- 3 files changed, 66 insertions(+), 5 deletions(-) create mode 100644 docs/reports/2026-10-11/service-cursor-hover.md diff --git a/docs/15-implementation-plan.md b/docs/15-implementation-plan.md index 308be23..55d342c 100644 --- a/docs/15-implementation-plan.md +++ b/docs/15-implementation-plan.md @@ -32,8 +32,9 @@ The shared-index/outline bundle is the next isolated recovery against merged but mixed retained allocation rises 7.5%, chiefly the cached outline. An explicit acceptance of that new retention trade-off is recorded; the scope, measurements and ownership evidence are in the [shared-index report](reports/2026-10-10/service-shared-indices.md). -Remaining candidates are cursor-local source lookup and accurate authored section -ranges, each measured independently. +Cursor-local source lookup replaces hover's whole-file key index +([cursor hover report](reports/2026-10-11/service-cursor-hover.md)). Accurate +authored section ranges remain, measured independently. The record-backed representation replacement remains parked; its recurring rebuild and first-use costs are recorded in the [service cost review](reports/2026-10-10/service-authoring-cost-review.md). @@ -88,7 +89,8 @@ the media-type fixes and the type-walk work landed together: [integration report](reports/2026-10-10/service-baseline-integration.md). Model-backed declaration inlays remove that first-use work from hints alone; their accepted request-order peak trade-off is recorded in the inlay recovery - report. Cursor-local source lookup remains a separately measured candidate. + report. Hover reads source keys along the cursor's path and keeps no index of + them. ## 3. Potential future work diff --git a/docs/reports/2026-10-11/service-cursor-hover.md b/docs/reports/2026-10-11/service-cursor-hover.md new file mode 100644 index 0000000..c506d2c --- /dev/null +++ b/docs/reports/2026-10-11/service-cursor-hover.md @@ -0,0 +1,54 @@ +# Cursor-local hover source lookup + +Date: 2026-10-11. Branch: `fix/parked-transfers`, from master `069b41f`; +measured against its parent `131b08e`. This is the third isolated experiment +in the [recovery plan](../../research/2026-10-10/service-recovery-plan.md). + +## 1. Goal, workload and cache boundaries + +The goal is to remove the whole-file source-key index from the first hover in +a file, without slowing repeated hovers or changing which token is explained. + +Before: the first hover in a URI sorted every key `syntax.keys` yields and +parsed every type value for built-in tokens, checked against a split of the +file's text. Both lists lived as long as the snapshot: 15,203 keys and about +2.6 MB on the hover corpus (one 13,204-line file). Later hovers bisected them. + +After: each hover reads `syntax.keys_at`, the keys enclosing the cursor, one +per mapping level, found by binary search over that mapping's entries, which +are in document order. Sequences are scanned, because an alias item keeps its +anchor's earlier position. Only the type value the cursor is in is parsed for +built-in tokens. Nothing is cached; the composed tree is still the snapshot's +or the shared source owner's (docs/21 § 4), so the cost of composition is +unchanged. Include-context discovery for headerless files is unchanged. + +A token's offset is taken as its column where the scalar's one-line span is +its text, or its text in quotes, rather than by comparing a line of the source. +An escape or a tag lengthens the span, so those tokens stay unexplained, as +before. A type expression folded over several lines now gets no token help; +before, a token on its first line could. + +## 2. Correctness and reach + +Over every TCK `.raml` file, `keys_at` at each key's start returns that key +with the same site and table as the whole-file walk (more than 10,000 keys). +A quoted union explains `nil` at its exact span; an escaped `n\x69l` explains +nothing. The existing hover suite, including unapplied-template built-ins, +opaque data, expanded aliases and inherited include contexts, passes +unchanged. The inlay reach test now fails if `Hover._keys_at` is called. + +## 3. Results + +`python -m bench ab 131b08e --bench hover --bench service-session --config unwrap --rounds 5`: + +| workload | time | noise | peak | retained | +|---|---|---|---|---| +| `hover` | 332.9 → 337.3 ms, noise | 16.2 % | unchanged | 26.29 → 23.71 MB (−9.8 %) | +| `service-session` | 1260.5 → 1142.8 ms (−9.3 %) | 7.4 % | unchanged | 27.50 → 24.87 MB (−9.6 %) | + +`hover` runs one probe per declaration, so its time is dominated by repeated +hovers; the removed index was paid once. `service-session` hovers sparsely +after each of three edits, so it pays the index once per edit and gains. + +`python -m bench linearity --bench hover`: time 0.992, peak 0.992, +retained 0.998. diff --git a/docs/research/2026-10-10/service-recovery-plan.md b/docs/research/2026-10-10/service-recovery-plan.md index 5f9532c..72d4d2d 100644 --- a/docs/research/2026-10-10/service-recovery-plan.md +++ b/docs/research/2026-10-10/service-recovery-plan.md @@ -143,8 +143,13 @@ Current execution: The implemented bundle passes local correctness/reach/scaling checks and improves sparse navigation 17.0%; its 7.5% mixed retained-allocation increase is attributed mainly to the requested-file outline cache. The full bundle is accepted with - that retention trade-off, subject to the normal CI/platform gate. - Remaining candidates stay separate experiments against the integrated baseline. + that retention trade-off. All CI checks passed; PR #4 merged at `069b41f`. +- Stage C, third bundle: cursor-local source lookup on `fix/parked-transfers` + from `069b41f`. Hover reads the keys along the cursor's path instead of a + whole-file index; mixed sessions are 9.3% faster and retain 9.6% less, with + results in the [cursor hover report](../../reports/2026-10-11/service-cursor-hover.md). + Accurate authored section ranges remain, recorded by the parser so the model + stays self-contained. Update these entries as each stage completes. Keep measurements in the dated report, current pending work in docs/15, and execution history here. From 6f5d091b4b5b54b4c29839f659cb8f383a03a6e0 Mon Sep 17 00:00:00 2001 From: Yury Date: Sun, 11 Oct 2026 01:26:07 +0300 Subject: [PATCH 8/9] feat: place outline sections at the keys their owners wrote The model kept no position for a section key, so the outline spanned a section's entries and selected the first, omitted an empty section, named a `schemas:` table `types`, and placed what an Extension restated under a master resource at its first entry. Decoders now record each section key an entity wrote in its own file in `Raml.written_sections`, by file: the root's tables and settings, a type's `facets:`, a resource's `uriParameters:`, a method's and a `describedBy:`'s parameter groups and `body:`, a response's `headers:` and `body:`. Each Overlay or Extension document's root keys and resource paths are recorded from its own tree before the merge, which keeps the master's key. A key is recorded only inside its owner's span, and not under a method or response a template wrote, which no outline lists. `bench ab HEAD --config unwrap`, 5-7 rounds: time within noise on endpoints, templates, large, service-navigation and service-session. Retained: endpoints 17.48 -> 17.77 MB (+1.65 %, inside the reported noise; 2,000 response `body:` keys), templates unchanged, large +0.3 %, service-navigation +0.5 %, service-session +0.3 %. Co-Authored-By: Claude Opus 5.5 --- docs/13-public-api.md | 5 + docs/21-language-service.md | 27 +++-- fastraml/parser/extensions.py | 18 +++ fastraml/parser/fragments.py | 32 +++++ fastraml/parser/security.py | 7 +- fastraml/parser/source_decode.py | 55 +++++++-- fastraml/registry.py | 43 ++++++- fastraml/service/outline.py | 181 +++++++++++++++++++---------- fastraml/sourceinfo.py | 34 ++++-- fastraml/types/shape.py | 1 + fastraml/yamlnode.py | 10 ++ tests/unit/test_service_queries.py | 100 +++++++++++++++- 12 files changed, 414 insertions(+), 99 deletions(-) diff --git a/docs/13-public-api.md b/docs/13-public-api.md index d8e466d..7af0955 100644 --- a/docs/13-public-api.md +++ b/docs/13-public-api.md @@ -136,6 +136,11 @@ Default lint no longer needs whole trees merely to detect `schema:` and retention. A consumer needing only that compact fact should read it rather than retaining the whole source tree. +`Raml.written_sections` holds, per file, each section key an entity wrote +there (`types:`, a method's `headers:`, a type's `facets:`), with its owner's +ID and where its value ends, also independent of source retention. A section +a template contributed is not recorded (docs/21 § 4). + A configuration file's `parser:` section is `ParserConfig`. `limits(options)` applies its `max_include_size`, `max_depth` and `regex_engine`; its `workspace_root` and `remote` are left to the host, which weighs them against diff --git a/docs/21-language-service.md b/docs/21-language-service.md index 81091fc..89178b8 100644 --- a/docs/21-language-service.md +++ b/docs/21-language-service.md @@ -197,10 +197,18 @@ The first outline request for a URI caches its complete result on the snapshot, including an empty outline. Later requests borrow the same list and symbols; callers must treat them as read-only. Root/dependency edits create a new snapshot and outline cache. A held older snapshot continues to answer from its older model. -Caching does not add authored section positions or populate source grammar. - -The model keeps no position for a section's key (`types:`, a method's -`headers:`), so a section spans its entries and selects the first. +Caching does not populate source grammar. + +A section is placed at the key its owner wrote (`types:`, a method's +`headers:`), spanning the key and its value, and is listed even when empty. +The decoders record these keys in `Raml.written_sections` (docs/13 § 2): the +root's, a type's `facets:`, a resource's `uriParameters:`, a method's or a +`describedBy:`'s `headers:`, `queryParameters:` and `body:`, a response's +`headers:` and `body:`, and a scheme's `describedBy:`. A key is recorded only +inside its owner's span in its file, and not under a method or response a +template wrote, which no outline lists. A table written `schemas:` is named +so. A section the parser recorded no key for spans its entries and selects +the first. What each entry lists is the authorship view's (`docs/16` § 10): a type's own members, not those it inherits, so an inherited property is outlined under the @@ -216,8 +224,10 @@ inside a resource or method (`docs/16` § 10). A file outlines what it wrote, selected by `location` over the model its snapshot parsed. An Extension or Overlay lists the types and other declarations it added to the master's tables, and, under a master resource's -path, the methods and resources it added there: a section spanning them, since -the resource's key it wrote is not in the model. The master lists its own. +path, the methods and resources it added there. The merge keeps the master's +key where both wrote one, so each document's root section keys and resource +paths are recorded from its own tree before the merge: the Extension's +`types:` and restated `/a:` are placed at its keys. The master lists its own. A trait or resource type is listed by name alone. Its body is decoded only where it is applied (`docs/08` § 5), and the model keeps it undecoded, so there @@ -516,8 +526,9 @@ written. It is the preview's source in `contrib/fastraml-vscode` lifetime, retained original trees and distinct JSON-include normalization. - `test_service_queries.py`: each query on one document with a library, a DataType include, a trait and a resource type; every query on a parse - stopped at each stage; an Extension's outline; and, over the TCK, that every - outline entry holds its selection and lies in its parent. + stopped at each stage; an Extension's outline and its own section keys; + empty, `schemas:` and template-supplied sections; and, over the TCK, that + every outline entry holds its selection and lies in its parent. - `test_service_hover.py`: contextual field meanings, full Markdown prose, authored and inherited summaries, aliases, optional versus nullable values, source-only template help, token boundaries, opaque data, educational examples, diff --git a/fastraml/parser/extensions.py b/fastraml/parser/extensions.py index 9d4142f..0a20b40 100644 --- a/fastraml/parser/extensions.py +++ b/fastraml/parser/extensions.py @@ -34,6 +34,7 @@ ReferenceResolver, identify_fragment, load_fragment_text, + record_root_sections, resolve_uses, unmarshal_uses, ) @@ -82,6 +83,8 @@ def decode_extension_chain(raml: Raml, uri: str, kind: FragmentKind, text: str) accumulator.add(RamlError.new('title is required', root_api.uri, API_HEAD_SPAN)) target = root_api.root + # Each document's own: the merge keeps the master's key where two wrote one. + record_root_sections(raml, api, target) declared_by: dict[str, dict[str, int]] = {} fragments: list[ExtensionFragment] = [] for position, document in enumerate(documents, start=1): @@ -92,6 +95,8 @@ def decode_extension_chain(raml: Raml, uri: str, kind: FragmentKind, text: str) fragment.api = api fragments.append(_register(raml, fragment)) _decode_own_keys(raml, fragment, document, accumulator) + record_root_sections(raml, fragment, document.root) + _record_resources(raml, fragment, document.root, '') result = merge_extension( target, document.root, @@ -220,6 +225,19 @@ def _decode_own_keys(raml: Raml, fragment: ExtensionFragment, document: _Documen raml.pop_ctx() +def _record_resources(raml: Raml, fragment: ExtensionFragment, node: Node, path: str) -> None: + """The resource keys an Overlay or Extension wrote, named by full path: one + it restates is merged into the master's, whose key the model keeps. + """ + if node.kind is not NodeKind.MAPPING: + return + for key, value in pairs(node): + if key.value.startswith('/'): + full = path + key.value + raml.record_section(fragment, key, value, fragment.location, name=full) + _record_resources(raml, fragment, value, full) + + def _resolve_libraries( raml: Raml, api: APIFragment, fragments: list[ExtensionFragment], accumulator: Accumulator ) -> None: diff --git a/fastraml/parser/fragments.py b/fastraml/parser/fragments.py index 389dad1..711dad6 100644 --- a/fastraml/parser/fragments.py +++ b/fastraml/parser/fragments.py @@ -125,6 +125,7 @@ 'parse_fragment', 'parse_included_fragment', 'parse_library', + 'record_root_sections', 'resolve_uses', ] @@ -1276,6 +1277,36 @@ def parse_fragment(raml: Raml, uri: str, kind: FragmentKind) -> Fragment: return decode_fragment(raml, uri, authored_kind, text) +#: The root keys that open a section of declarations or settings (docs/21 § 4). +_ROOT_SECTIONS: Final = frozenset( + { + FACET_USES, + FACET_TYPES, + FACET_SCHEMAS, + FACET_ANNOTATION_TYPES, + FACET_TRAITS, + FACET_RESOURCE_TYPES, + FACET_SECURITY_SCHEMES, + FACET_BASE_URI_PARAMETERS, + FACET_DOCUMENTATION, + } +) +_DOCUMENT_KINDS: Final = frozenset( + {FragmentKind.API, FragmentKind.LIBRARY, FragmentKind.OVERLAY, FragmentKind.EXTENSION} +) + + +def record_root_sections(raml: Raml, fragment: Fragment, root: Node) -> None: + """The section keys `fragment`'s own document wrote at its root, before + any merge: a fragment that is one declaration has `uses:` only. + """ + if root.kind is not NodeKind.MAPPING: + return + for key, value in pairs(root): + if key.value == FACET_USES or (key.value in _ROOT_SECTIONS and fragment.kind in _DOCUMENT_KINDS): + raml.record_section(fragment, key, value, fragment.location) + + def decode_fragment(raml: Raml, uri: str, kind: FragmentKind, text: str) -> Fragment: """Register, decode, then resolve `uses:` — in that order. See the module docstring.""" raml.store_source_text(uri, text) @@ -1302,6 +1333,7 @@ def decode_fragment(raml: Raml, uri: str, kind: FragmentKind, text: str) -> Frag try: root = compose(text, uri=uri, max_depth=raml.max_depth, key_pool=raml.mapping_keys) raml.store_source_node(uri, root) + record_root_sections(raml, fragment, root) fragment.decode(root) except RamlError as err: # The `uses:` that decoded are still resolved, so a mistake in the body diff --git a/fastraml/parser/security.py b/fastraml/parser/security.py index d7baec8..22ed312 100644 --- a/fastraml/parser/security.py +++ b/fastraml/parser/security.py @@ -29,6 +29,8 @@ FACET_DESCRIBED_BY, FACET_DESCRIPTION, FACET_DISPLAY_NAME, + FACET_HEADERS, + FACET_QUERY_PARAMETERS, FACET_REQUEST_TOKEN_URI, FACET_RESPONSES, FACET_SCOPES, @@ -223,6 +225,7 @@ def make_security_scheme_definition( # noqa: PLR0912 - one pass over the declar elif name == FACET_DESCRIPTION: definition.description = make_string_facet(raml, key, value, location) elif name == FACET_DESCRIBED_BY: + raml.record_section(definition, key, value, location) _decode_described_by(raml, value, location, partial(setattr, definition, 'described_by')) elif name == FACET_SETTINGS: settings_node = value @@ -263,11 +266,13 @@ def _decode_described_by( accumulator = Accumulator() for key, value in pairs(node): name = key.value + if name in (FACET_HEADERS, FACET_QUERY_PARAMETERS): + raml.record_section(description, key, value, location) try: if decode_request_facet(raml, description, key, value, location): continue if name == FACET_RESPONSES: - decode_responses(raml, value, location, description.responses) + decode_responses(raml, value, location, description.responses, holder=description) elif is_annotation_key(name): add_domain_extension(raml, description.annotations, location, key, value) else: diff --git a/fastraml/parser/source_decode.py b/fastraml/parser/source_decode.py index 9560569..d22e9d2 100644 --- a/fastraml/parser/source_decode.py +++ b/fastraml/parser/source_decode.py @@ -45,6 +45,7 @@ from fastraml.parser.facets import MEDIA_RANGE, make_string_facet from fastraml.parser.includes import inline_include from fastraml.parser.syntax import is_media_type_map as _is_media_type_map +from fastraml.registry import written_inside from fastraml.types.shape import make_body_shape, make_parameter_map, make_shape from fastraml.yamlnode import NodeKind, is_null, node_error, pairs @@ -57,7 +58,13 @@ from fastraml.types.base import BaseShape, Parameter from fastraml.yamlnode import Node -__all__ = ['decode_request_facet', 'decode_responses', 'decode_source_endpoint', 'query_exclusion_error'] +__all__ = [ + 'REQUEST_SECTIONS', + 'decode_request_facet', + 'decode_responses', + 'decode_source_endpoint', + 'query_exclusion_error', +] @contextmanager @@ -181,12 +188,16 @@ def _is_status_code(value: str) -> bool: return _STATUS_CODE.match(value) is not None -def _decode_response(raml: Raml, key: Node, value: Node, location: str, attach: Callable[[Response], None]) -> None: +def _decode_response( # noqa: PLR0913 - the response's pair, where it goes, and its holder + raml: Raml, key: Node, value: Node, location: str, attach: Callable[[Response], None], *, holder: object | None +) -> None: """One response, attached before its content is decoded. A response whose content fails stays attached, marked in `Raml.broken` - (docs/13 § 1). + (docs/13 § 1). Its section keys are recorded only where `holder` wrote + it, not a template. """ + written = holder is not None and written_inside(holder, key) location = raml.location_of(value, location) response = Response( id=raml.next_id(), @@ -208,6 +219,8 @@ def _decode_response(raml: Raml, key: Node, value: Node, location: str, attach: with raml.target_scope(DomainLocation.RESPONSE): for child_key, child_value in pairs(value): name = child_key.value + if written and name in (FACET_HEADERS, FACET_BODY): + raml.record_section(response, child_key, child_value, location) try: if name == FACET_DISPLAY_NAME: response.display_name = make_string_facet(raml, child_key, child_value, location) @@ -228,9 +241,12 @@ def _decode_response(raml: Raml, key: Node, value: Node, location: str, attach: accumulator.raise_if_any() -def decode_responses(raml: Raml, node: Node, location: str, responses: dict[str, Response]) -> None: +def decode_responses( + raml: Raml, node: Node, location: str, responses: dict[str, Response], *, holder: object | None +) -> None: """A `responses:` map, into the holder's own. Public because `describedBy:` - reuses it verbatim. + reuses it verbatim. `holder` is the entity that wrote it, or `None` for + one a template wrote, whose section keys are not recorded. """ node, location = inline_include(raml, node, location) if is_null(node): @@ -243,7 +259,7 @@ def decode_responses(raml: Raml, node: Node, location: str, responses: dict[str, try: if not _is_status_code(key.value): raise node_error('status code must be a 3-digit number', location, key, info={'code': key.value}) - _decode_response(raml, key, value, location, partial(setitem, responses, key.value)) + _decode_response(raml, key, value, location, partial(setitem, responses, key.value), holder=holder) except RamlError as err: accumulator.add(err) accumulator.raise_if_any() @@ -260,6 +276,10 @@ class RequestFacets(Protocol): query_string: BaseShape | None +#: The keys of a method that open a section (docs/21 § 4). +REQUEST_SECTIONS: Final = frozenset({FACET_HEADERS, FACET_QUERY_PARAMETERS, FACET_BODY}) + + def decode_request_facet(raml: Raml, into: RequestFacets, key: Node, value: Node, location: str) -> bool: """`headers`, `queryParameters` or `queryString`. Returns whether it was one.""" name = key.value @@ -290,8 +310,10 @@ def query_exclusion_error(raml: Raml, facets: RequestFacets, location: str, node def _decode_operation_field( # noqa: PLR0913, PLR0917 - one pass over the method's key vocabulary - raml: Raml, operation: Operation, request: Request, key: Node, value: Node, location: str + raml: Raml, operation: Operation, request: Request, key: Node, value: Node, location: str, *, written: bool ) -> None: + if written and key.value in REQUEST_SECTIONS: + raml.record_section(operation, key, value, location) if decode_request_facet(raml, request, key, value, location): return name = key.value @@ -304,18 +326,22 @@ def _decode_operation_field( # noqa: PLR0913, PLR0917 - one pass over the metho elif name == FACET_BODY: _decode_bodies(raml, key, value, location, DomainLocation.REQUEST_BODY, request.bodies) elif name == FACET_RESPONSES: - decode_responses(raml, value, location, operation.responses) + decode_responses(raml, value, location, operation.responses, holder=operation if written else None) elif is_annotation_key(name): add_domain_extension(raml, operation.annotations, location, key, value) else: raml.recover(node_error('unknown field', location, key, info={'field': name})) -def decode_source_operation(raml: Raml, source: SourceOperation, attach: Callable[[Operation], None]) -> None: +def decode_source_operation( + raml: Raml, source: SourceOperation, attach: Callable[[Operation], None], *, holder: EndPoint +) -> None: """One method's retained tree into an `Operation`, attached before its content is decoded. One whose content fails stays attached, - marked in `Raml.broken` (docs/13 § 1). + marked in `Raml.broken` (docs/13 § 1). Its section keys are recorded only + where `holder` wrote it, not a resource type. """ + written = written_inside(holder, source.key_pos) operation = Operation( id=raml.next_id(), method=source.method, @@ -349,7 +375,9 @@ def decode_source_operation(raml: Raml, source: SourceOperation, attach: Callabl # survive the containers the merge synthesised. A pair a # template grafted is located where the template was written. with raml.provenance_scope(value): - _decode_operation_field(raml, operation, request, key, value, raml.location_of(key, location)) + _decode_operation_field( + raml, operation, request, key, value, raml.location_of(key, location), written=written + ) except RamlError as err: accumulator.add(err) accumulator.add(query_exclusion_error(raml, request, location, source.body)) @@ -368,6 +396,7 @@ def _decode_endpoint_field(raml: Raml, endpoint: EndPoint, key: Node, value: Nod elif name == FACET_DESCRIPTION: endpoint.description = make_string_facet(raml, key, value, location) elif name == FACET_URI_PARAMETERS: + raml.record_section(endpoint, key, value, location) make_parameter_map(raml, value, location, 'uri', endpoint.uri_parameters) elif is_annotation_key(name): add_domain_extension(raml, endpoint.annotations, location, key, value) @@ -420,7 +449,9 @@ def decode_source_endpoint(raml: Raml, source: SourceEndPoint, attach: Callable[ for method, operation_source in source.operations.items(): try: - decode_source_operation(raml, operation_source, partial(setitem, endpoint.operations, method)) + decode_source_operation( + raml, operation_source, partial(setitem, endpoint.operations, method), holder=endpoint + ) except RamlError as err: accumulator.add(err) diff --git a/fastraml/registry.py b/fastraml/registry.py index d02bb72..a63dc3b 100644 --- a/fastraml/registry.py +++ b/fastraml/registry.py @@ -23,8 +23,9 @@ from fastraml.domains import DomainLocation from fastraml.errors import Accumulator, RamlError from fastraml.loaders import SchemeLoader -from fastraml.sourceinfo import KeywordUse -from fastraml.yamlnode import AUTHORED_NODES, DEFAULT_MAX_DEPTH, mark_subtree +from fastraml.positions import UNKNOWN, Position +from fastraml.sourceinfo import KeywordUse, WrittenSection +from fastraml.yamlnode import AUTHORED_NODES, DEFAULT_MAX_DEPTH, mark_subtree, written_end if TYPE_CHECKING: from collections.abc import Iterator, Mapping, Sequence @@ -36,7 +37,6 @@ from fastraml.parser.includes import IncludeRef from fastraml.parser.structural_merge import ProvenanceOverlay from fastraml.parser.substitutions import Substitutions - from fastraml.positions import Position from fastraml.types.base import Property from fastraml.types.expressions import ExprCache from fastraml.types.schema_compile import SchemaRegistry @@ -79,6 +79,24 @@ class Stage(Enum): VALIDATED = 'validated' # P10, when requested +def written_inside(owner: object, node: Node | Position) -> bool: + """Whether `node` starts inside `owner`'s span, from its key through its + value; true for an owner placed nowhere. Compared as numbers: a parse + asks once per section key. + """ + key: Position = getattr(owner, 'key_pos', UNKNOWN) + value: Position = getattr(owner, 'value_pos', UNKNOWN) + if not key.is_known: + key = value + if not value.is_known: + value = key + if not key.is_known: + return True + start = min((key.line, key.column), (value.line, value.column)) + end = max((key.end_line, key.end_column), (value.end_line, value.end_column)) + return start <= (node.line, node.column) <= end + + class Identified(Protocol): """Anything `Raml.broken` can mark: every model entity has an id.""" @@ -213,6 +231,7 @@ class Raml: 'shapes', 'substitutions', 'syntax_aliases', + 'written_sections', # --- work queues ----------------------------------------------------- '_discriminator_shapes', 'unresolved_shapes', @@ -302,6 +321,9 @@ def __init__( # noqa: PLR0913 - the parse's configuration, keyword-only, one fi #: Accepted compatibility spellings by entity ID; only used spellings #: allocate a record. The parser records syntax, a view judges it. self.syntax_aliases: dict[int, list[KeywordUse]] = {} + #: The section keys entities wrote, by the file they wrote them in, in + #: the order decoded (docs/21 § 4). + self.written_sections: dict[str, list[WrittenSection]] = {} # A worklist, drained from the left in P7 while resolution appends to # the right; a deque keeps both ends O(1). @@ -644,6 +666,21 @@ def record_syntax_alias(self, entity: Identified, key: Node, location: str) -> N use = KeywordUse(self.location_of(key, location), key.position, key.value) self.syntax_aliases.setdefault(entity.id, []).append(use) + def record_section(self, owner: Identified, key: Node, value: Node, location: str, *, name: str = '') -> None: + """Where `owner` wrote a section key (docs/21 § 4). + + Only in `owner`'s file and, where it is placed, inside its span: a + section a template grafted was written in the template. `name`, when + given, replaces the key's: a resource's full path. + """ + if not written_inside(owner, key): + return + where = self.location_of(key, location) + if where != getattr(owner, 'location', where): + return + section = WrittenSection(owner.id, name or key.value, key.position, *written_end(value)) + self.written_sections.setdefault(where, []).append(section) + def put_source_info(self, entity_id: int, key: Node | None, value: Node) -> None: """Index an entity's authored nodes when source retention is on.""" if self.source_info is not None: diff --git a/fastraml/service/outline.py b/fastraml/service/outline.py index 1aa708c..9b79a34 100644 --- a/fastraml/service/outline.py +++ b/fastraml/service/outline.py @@ -24,12 +24,33 @@ from fastraml.parser.documentation import DocumentationItem from fastraml.parser.endpoints import Body, Operation, Request, Response from fastraml.parser.templates import TemplateDefinition + from fastraml.registry import Identified, Raml from fastraml.service.workspace import Snapshot from fastraml.types.base import Parameter, ScalarFacet __all__ = ['document_symbols'] +class _Written: + """The section keys entities wrote in one file (`Raml.written_sections`), + by owner, indexed once per outline. + """ + + __slots__ = ('_by_owner',) + + def __init__(self, raml: Raml, uri: str) -> None: + self._by_owner: dict[int, dict[str, _Placed]] = {} + for each in raml.written_sections.get(uri, ()): + self._by_owner.setdefault(each.owner, {}).setdefault(each.name, (uri, each.span, each.key)) + + def of(self, owner: Identified) -> Mapping[str, _Placed]: + return self._by_owner.get(owner.id, {}) + + +#: Where a section key was written: its file, its span and the key's own span. +type _Placed = tuple[str, Position, Position] + + def document_symbols(snapshot: Snapshot, uri: str) -> list[Symbol]: """Borrow the read-only outline cached for this URI and snapshot.""" if uri not in snapshot.outlines: @@ -43,8 +64,8 @@ def _document_symbols(snapshot: Snapshot, uri: str) -> list[Symbol]: Declarations only, as a code outline lists them: a type and its members, not its examples or annotations. A type's detail is its type as written, - and an optional member is named `name?`. A section the model records no - key for, `types:` or a method's `headers:`, spans its entries. + and an optional member is named `name?`. A section is placed at the key + its owner wrote, which the parser records, and listed even when empty. What an Extension or an Overlay adds is outlined in it: a type it declared in the master's table, and under a master resource's path, a method it @@ -54,30 +75,37 @@ def _document_symbols(snapshot: Snapshot, uri: str) -> list[Symbol]: semantic = snapshot.semantic if raml is None or semantic is None or uri not in raml.fragments: return [] + written = _Written(raml, uri) + root = written.of(raml.fragments[uri]) found: list[Symbol | None] = [ symbol(name, SymbolKind.METADATA, facet, facet.value) for name, facet in authored.metadata(raml, uri) ] - parameters = (_parameter(name, param) for name, param in authored.base_uri_parameters(raml, uri)) - found.append(_group('baseUriParameters', parameters)) - found.append(_group('documentation', map(_documentation, authored.documentation(raml, uri)))) + parameters = (_parameter(name, param, written) for name, param in authored.base_uri_parameters(raml, uri)) + found.append(_group('baseUriParameters', parameters, root)) + found.append(_group('documentation', map(_documentation, authored.documentation(raml, uri)), root)) uses = (symbol(name, SymbolKind.LIBRARY, link, link.value) for name, link in authored.uses(raml, uri).items()) - found.append(_group('uses', uses)) - sections: dict[str, list[Symbol | None]] = {} + found.append(_group('uses', uses, root)) + sections: dict[str, list[Symbol | None]] = {key: [] for key in DECLARATION_KINDS if _spelled(key, root) in root} for key, name, entity in semantic.by_uri.get(uri, ()): - sections.setdefault(key, []).append(_declaration(name, DECLARATION_KINDS[key], entity)) - found += (_group(key, entries) for key, entries in sections.items()) - found += (_resource(written) for written in authored.resources(raml, uri)) - found += _fragment_body(authored.fragment_body(raml, uri)) + sections.setdefault(key, []).append(_declaration(name, DECLARATION_KINDS[key], entity, written)) + found += (_group(_spelled(key, root), entries, root) for key, entries in sections.items()) + found += (_resource(each, written, root) for each in authored.resources(raml, uri)) + found += _fragment_body(authored.fragment_body(raml, uri), written) return _here(uri, found) -def _fragment_body(body: BaseShape | SecuritySchemeDefinition | None) -> list[Symbol | None]: +def _spelled(key: str, written: Mapping[str, _Placed]) -> str: + """The table's key as the file wrote it: `schemas` is `types` in the model.""" + return 'schemas' if key == 'types' and 'schemas' in written and 'types' not in written else key + + +def _fragment_body(body: BaseShape | SecuritySchemeDefinition | None, written: _Written) -> list[Symbol | None]: """What a fragment file that is one declaration wrote, at the top: the file is the declaration, and its name is the file's. """ if isinstance(body, BaseShape): - return _member_symbols(body) - return [] if body is None else [_described(body)] + return _member_symbols(body, written) + return [] if body is None else [_described(body, written)] def _documentation(item: DocumentationItem) -> Symbol | None: @@ -86,61 +114,61 @@ def _documentation(item: DocumentationItem) -> Symbol | None: return None if title is None else symbol(str(title.value), SymbolKind.DOCUMENTATION, item, key=title.value_pos) -def _declaration(name: str, kind: SymbolKind, entity: object) -> Symbol | None: +def _declaration(name: str, kind: SymbolKind, entity: object, written: _Written) -> Symbol | None: if isinstance(entity, BaseShape): - return _type(name, kind, entity) + return _type(name, kind, entity, written) if isinstance(entity, SecuritySchemeDefinition): found = symbol(name, kind, entity, entity.resolved().type) if found is not None: - _adopt(found, [_described(entity)]) + _adopt(found, [_described(entity, written)]) return found return symbol(name, kind, cast('TemplateDefinition', entity)) -def _type(name: str, kind: SymbolKind, base: BaseShape) -> Symbol | None: +def _type(name: str, kind: SymbolKind, base: BaseShape, written: _Written) -> Symbol | None: """A type-like symbol: `base` and the members it declares.""" found = symbol(name, kind, base, _written(base)) if found is None: return None found.form = 'enum' if base.enum is not None else base.type or '' - _members(base, found) + _adopt(found, _member_symbols(base, written)) return found -def _described(definition: SecuritySchemeDefinition) -> Symbol | None: +def _described(definition: SecuritySchemeDefinition, written: _Written) -> Symbol | None: """A security scheme's `describedBy`, as the definition wrote it.""" described = definition.described_by - return None if described is None else _group('describedBy', _message(described, definition)) - - -def _members(base: BaseShape, parent: Symbol) -> None: - """Add to `parent` the members `base` declares.""" - _adopt(parent, _member_symbols(base)) + if described is None: + return None + children = _message(described, definition, written, written.of(described)) + return _group('describedBy', children, written.of(definition)) -def _member_symbols(base: BaseShape) -> list[Symbol | None]: +def _member_symbols(base: BaseShape, written: _Written) -> list[Symbol | None]: """The members `base` declares: not those it inherits, nor `items` an expression built (docs/16 § 10). """ found: list[Symbol | None] = [ - _type(key if prop.required else f'{key}?', SymbolKind.PROPERTY, prop.base) + _type(key if prop.required else f'{key}?', SymbolKind.PROPERTY, prop.base, written) for key, prop in authored.properties(base) ] found += ( - _type(f'/{key}/', SymbolKind.PROPERTY, pattern.base) for key, pattern in authored.pattern_properties(base) + _type(f'/{key}/', SymbolKind.PROPERTY, pattern.base, written) + for key, pattern in authored.pattern_properties(base) ) items = authored.items(base) if items is not None: - found.append(_type('items', SymbolKind.TYPE, items)) + found.append(_type('items', SymbolKind.TYPE, items, written)) facets = ( - _type(key if prop.required else f'{key}?', SymbolKind.FACET, prop.base) for key, prop in authored.facets(base) + _type(key if prop.required else f'{key}?', SymbolKind.FACET, prop.base, written) + for key, prop in authored.facets(base) ) - found.append(_group('facets', facets)) + found.append(_group('facets', facets, written.of(base))) return found -def _parameter(name: str, param: Parameter) -> Symbol | None: - return _type(name if param.required else f'{name}?', SymbolKind.PARAMETER, param.declaration.base) +def _parameter(name: str, param: Parameter, written: _Written) -> Symbol | None: + return _type(name if param.required else f'{name}?', SymbolKind.PARAMETER, param.declaration.base, written) def _written(base: BaseShape) -> str: @@ -156,77 +184,92 @@ def _written(base: BaseShape) -> str: return written.value.lstrip() -def _resource(written: authored.WrittenResource) -> Symbol | None: +def _resource(resource: authored.WrittenResource, written: _Written, root: Mapping[str, _Placed]) -> Symbol | None: """A resource as this file wrote it: at its key when the file declares - it, or else a section, named by its path, holding what the file added. + it, or else at the key it restated it under, holding what it added. """ - endpoint = written.endpoint - children: list[Symbol | None] = [_method(operation) for operation in written.operations] - children += (_resource(child) for child in written.resources) - if not written.here: - return _group(endpoint.uri, children, SymbolKind.RESOURCE) + endpoint = resource.endpoint + children: list[Symbol | None] = [_method(operation, written) for operation in resource.operations] + children += (_resource(child, written, root) for child in resource.resources) + if not resource.here: + # An Overlay or Extension recorded the path it restated (docs/21 § 4). + return _group(endpoint.uri, children, root, SymbolKind.RESOURCE, key=endpoint.full_uri) found = symbol(endpoint.uri, SymbolKind.RESOURCE, endpoint, _shown(endpoint.display_name)) if found is None: return None applied = () if endpoint.resource_type is None else (endpoint.resource_type,) - parameters = (_parameter(name, param) for name, param in authored.parameters(endpoint, endpoint.uri_parameters)) + parameters = ( + _parameter(name, param, written) for name, param in authored.parameters(endpoint, endpoint.uri_parameters) + ) _adopt( found, [ _applied('type', authored.members(endpoint, applied)), _applied('is', authored.members(endpoint, endpoint.traits)), _applied('securedBy', authored.secured_by(endpoint)), - _group('uriParameters', parameters), + _group('uriParameters', parameters, written.of(endpoint)), *children, ], ) return found -def _method(operation: Operation) -> Symbol | None: +def _method(operation: Operation, written: _Written) -> Symbol | None: found = symbol(operation.method, SymbolKind.METHOD, operation, _shown(operation.display_name)) if found is None: return None + sections = written.of(operation) children: list[Symbol | None] = [ _applied('is', authored.members(operation, operation.traits)), _applied('securedBy', authored.secured_by(operation)), ] if operation.request is not None: - children += _message(operation.request, operation) - children.append(_group('body', _bodies(operation.request.bodies, operation))) - children += (_response(response) for response in authored.members(operation, operation.responses.values())) + children += _message(operation.request, operation, written, sections) + children.append(_group('body', _bodies(operation.request.bodies, operation, written), sections)) + children += (_response(response, written) for response in authored.members(operation, operation.responses.values())) _adopt(found, children) return found def _message( - message: Request | SecuritySchemeDescription, owner: Operation | SecuritySchemeDefinition + message: Request | SecuritySchemeDescription, + owner: Operation | SecuritySchemeDefinition, + written: _Written, + sections: Mapping[str, _Placed], ) -> list[Symbol | None]: """A request's, or a `describedBy`'s, parameter groups and query string, - as `owner` wrote them; a `describedBy`'s responses too. + as `owner` wrote them, with the section keys it recorded; a `describedBy`'s + responses too. """ - query = (_parameter(name, param) for name, param in authored.parameters(owner, message.query_parameters)) - headers = (_parameter(name, param) for name, param in authored.parameters(owner, message.headers)) - found: list[Symbol | None] = [_group('queryParameters', query), _group('headers', headers)] + query = (_parameter(name, param, written) for name, param in authored.parameters(owner, message.query_parameters)) + headers = (_parameter(name, param, written) for name, param in authored.parameters(owner, message.headers)) + found: list[Symbol | None] = [_group('queryParameters', query, sections), _group('headers', headers, sections)] string = message.query_string if string is not None and authored.wrote(owner, string.location, string.key_pos): - found.append(_type('queryString', SymbolKind.TYPE, string)) + found.append(_type('queryString', SymbolKind.TYPE, string, written)) if isinstance(message, SecuritySchemeDescription): - found += (_response(response) for response in authored.members(owner, message.responses.values())) + found += (_response(response, written) for response in authored.members(owner, message.responses.values())) return found -def _response(response: Response) -> Symbol | None: +def _response(response: Response, written: _Written) -> Symbol | None: shown = _shown(response.display_name) or _shown(response.description) found = symbol(response.code, SymbolKind.RESPONSE, response, shown) if found is None: return None - headers = (_parameter(name, param) for name, param in authored.parameters(response, response.headers)) - _adopt(found, [_group('headers', headers), _group('body', _bodies(response.bodies, response))]) + sections = written.of(response) + headers = (_parameter(name, param, written) for name, param in authored.parameters(response, response.headers)) + _adopt( + found, + [ + _group('headers', headers, sections), + _group('body', _bodies(response.bodies, response, written), sections), + ], + ) return found -def _bodies(bodies: Mapping[str, Body], owner: Operation | Response) -> Iterator[Symbol | None]: +def _bodies(bodies: Mapping[str, Body], owner: Operation | Response, written: _Written) -> Iterator[Symbol | None]: """One symbol per body `owner` wrote: a `body:` with no media type is one body per default media type (docs/08 § 6.3), named by all of them. """ @@ -236,7 +279,7 @@ def _bodies(bodies: Mapping[str, Body], owner: Operation | Response) -> Iterator # The body's own key: its shape's is none for `body: Book`. found = symbol(name, SymbolKind.BODY, body, '' if shape is None else _written(shape)) if found is not None and shape is not None: - _members(shape, found) + _adopt(found, _member_symbols(shape, written)) yield found @@ -254,12 +297,24 @@ def _applied(name: str, refs: Iterable[DirectiveRef | SecurityScheme]) -> Symbol ) -def _group(name: str, children: Iterable[Symbol | None], kind: SymbolKind = SymbolKind.SECTION) -> Symbol | None: - """A section holding `children`, in the order written, or `None` for an - empty one. The model keeps no position for a section's key, so it spans - its entries and selects the first. +def _group( + name: str, + children: Iterable[Symbol | None], + sections: Mapping[str, _Placed], + kind: SymbolKind = SymbolKind.SECTION, + *, + key: str = '', +) -> Symbol | None: + """A section holding `children`, in the order written, at the key its + owner wrote, `sections[key or name]`. One the parser recorded no key for, + such as a section a template supplied, spans its entries and selects the + first, or is `None` when empty. """ placed = _ordered(children) + at = sections.get(key or name) + if at is not None: + uri, span, selection = at + return Symbol(name, kind, uri, span, selection, placed) if not placed: return None first = placed[0] diff --git a/fastraml/sourceinfo.py b/fastraml/sourceinfo.py index 2e566be..5e5dca3 100644 --- a/fastraml/sourceinfo.py +++ b/fastraml/sourceinfo.py @@ -1,17 +1,16 @@ -"""Facts a decoder records when it accepts a compatibility spelling. +"""Facts a decoder records about how a document was written. -A `KeywordUse` names where a deprecated keyword was written. It is a model -fact the registry stores and a view may report on: not a diagnostic, and it -keeps no YAML node (docs/04 § 5, docs/05 § 3). +A `KeywordUse` names where a deprecated keyword was written; a `WrittenSection` +where an entity wrote a section key, such as `types:` or a method's `headers:`. +Each is a model fact the registry stores and a view may read: not a +diagnostic, and it keeps no YAML node (docs/04 § 5, docs/05 § 3, docs/21 § 4). """ from __future__ import annotations from dataclasses import dataclass -from typing import TYPE_CHECKING -if TYPE_CHECKING: - from fastraml.positions import Position +from fastraml.positions import Position @dataclass(frozen=True, slots=True, eq=False) @@ -21,3 +20,24 @@ class KeywordUse: location: str position: Position name: str + + +@dataclass(frozen=True, slots=True, eq=False) +class WrittenSection: + """A section key one entity wrote in its own file, and where the entry + it opens ends. Held per file, so the record carries its owner's ID. + + `name` is the key as written (`schemas`, not `types`), or for a resource + an Extension restated, the resource's full path. + """ + + owner: int + name: str + key: Position + end_line: int + end_column: int + + @property + def span(self) -> Position: + """From the key through the end of its value.""" + return Position(self.key.line, self.key.column, self.end_line, self.end_column) diff --git a/fastraml/types/shape.py b/fastraml/types/shape.py index 773caa2..9ddbda8 100644 --- a/fastraml/types/shape.py +++ b/fastraml/types/shape.py @@ -387,6 +387,7 @@ def _decode( # noqa: PLR0912 - one pass over the common-facet vocabulary (docs/ case fn.FACET_REQUIRED: base.required = make_bool_facet(raml, key, value, location) case fn.FACET_FACETS: + raml.record_section(base, key, value, location) _decode_custom_facet_defs(raml, base, value) case fn.FACET_EXAMPLE: _decode_example(raml, base, key, value) diff --git a/fastraml/yamlnode.py b/fastraml/yamlnode.py index 7108f48..357df32 100644 --- a/fastraml/yamlnode.py +++ b/fastraml/yamlnode.py @@ -50,6 +50,7 @@ 'with_content', 'with_grafts', 'with_value', + 'written_end', ] try: # pragma: no cover - depends on how PyYAML was built @@ -490,6 +491,15 @@ def _end(node: Node) -> tuple[int, int]: return end +def written_end(node: Node) -> tuple[int, int]: + """Where `full_position` ends, without building the `Position`.""" + if not node.content: + return node.end_line, node.end_column + if isinstance(node, _Grafted): + return node.written_end + return _end(node) + + def end_line(node: Node) -> int: """The last line `full_position` spans.""" return node.full_position.end_line diff --git a/tests/unit/test_service_queries.py b/tests/unit/test_service_queries.py index 15cd0a2..4ff224b 100644 --- a/tests/unit/test_service_queries.py +++ b/tests/unit/test_service_queries.py @@ -338,7 +338,8 @@ def outlined(name: str): ('401', SymbolKind.RESPONSE, ''), ])] # fmt: skip assert outlined('home.raml') == [('documentation', section, '', [('Home', SymbolKind.DOCUMENTATION, '')])] - assert [each[0] for each in outlined('api.raml')] == ['title', 'securitySchemes', 'types'] + # `documentation:` is written here, its one item in its own file. + assert [each[0] for each in outlined('api.raml')] == ['title', 'documentation', 'securitySchemes', 'types'] def test_a_documentation_item_file_no_api_reads_outlines_its_title(self, memory_workspace): files = {'home.raml': '#%RAML 1.0 DocumentationItem\ntitle: Home\ncontent: hi\n'} @@ -387,12 +388,101 @@ def test_a_file_outlines_what_it_wrote_an_extension_what_it_added(self, memory_w ('/a', resource, '', [('get', method, '')]), ] - def test_a_section_spans_its_entries_and_selects_the_first(self, parsed): + def test_a_section_spans_its_key_and_entries_and_selects_its_key(self, parsed): + # docs/21 § 4: the decoder records the key the owner wrote. snapshot, folder = parsed types = next(s for s in outline.document_symbols(snapshot, f'{folder}/api.raml') if s.name == 'types') - entity, book = types.children - assert (types.span.line, types.span.end_line) == (entity.span.line, book.span.end_line) - assert types.selection == entity.selection + _entity, book = types.children + line, column = _where(API, 'types:') + assert (types.span.line, types.span.column, types.span.end_line) == (line, column, book.span.end_line) + assert (types.selection.line, types.selection.column, types.selection.end_column) == (line, column, column + 5) + + def test_an_empty_or_deprecated_section_is_outlined_at_its_key(self, memory_workspace): + document = ( + '#%RAML 1.0\ntitle: T\nschemas:\n A: string\ntraits: {}\n' + '/r:\n uriParameters:\n get:\n headers:\n responses:\n 200:\n body:\n' + ) + workspace, folder = _buffered(memory_workspace, {'api.raml': document}) + snapshot = workspace.snapshot(f'{folder}/api.raml') + found = outline.document_symbols(snapshot, f'{folder}/api.raml') + section = SymbolKind.SECTION + assert _tree(found) == [ + ('title', SymbolKind.METADATA, 'T'), + ('schemas', section, '', [('A', SymbolKind.TYPE, 'string')]), + ('traits', section, ''), + ('/r', SymbolKind.RESOURCE, '', [ + ('uriParameters', section, ''), + ('get', SymbolKind.METHOD, '', [ + ('headers', section, ''), + ('200', SymbolKind.RESPONSE, '', [('body', section, '')]), + ]), + ]), + ] # fmt: skip + assert found[2].selection.line == _where(document, 'traits')[0] + + def test_a_section_a_template_supplied_is_not_the_methods(self, memory_workspace): + # The trait wrote `queryParameters:`; the method wrote `headers:` only. + document = ( + '#%RAML 1.0\ntitle: T\ntraits:\n paged:\n queryParameters:\n page: integer\n' + '/r:\n get:\n is: [paged]\n headers:\n X-Id: string\n' + ) + workspace, folder = _buffered(memory_workspace, {'api.raml': document}) + snapshot = workspace.snapshot(f'{folder}/api.raml') + resource = outline.document_symbols(snapshot, f'{folder}/api.raml')[-1] + (method,) = resource.children + assert [(child.name, child.selection.line) for child in method.children] == [ + ('is', _where(document, 'paged]')[0]), + ('headers', _where(document, 'headers:')[0]), + ] + + def test_the_parser_records_no_section_of_what_a_template_wrote(self, memory_workspace): + # The resource type's method and response are written in the template; + # recording their sections at each application would grow the model. + document = ( + '#%RAML 1.0\ntitle: T\nresourceTypes:\n rt:\n get:\n queryParameters:\n q: string\n' + ' responses:\n 200:\n body:\n application/json:\n' + '/a:\n type: rt\n/b:\n type: rt\n uriParameters: {}\n' + ) + workspace, folder = _buffered(memory_workspace, {'api.raml': document}) + raml = workspace.snapshot(f'{folder}/api.raml').raml + assert raml is not None + recorded = [(each.name, each.key.line) for each in raml.written_sections[f'{folder}/api.raml']] + assert recorded == [('resourceTypes', 3), ('uriParameters', _where(document, 'uriParameters')[0])] + + def test_a_types_facets_and_a_schemes_described_by_are_placed_at_their_keys(self, memory_workspace): + document = ( + '#%RAML 1.0\ntitle: T\ntypes:\n A:\n facets:\n unit: string\n' + 'securitySchemes:\n s:\n type: Basic Authentication\n describedBy:\n' + ' headers:\n Authorization: string\n' + ) + workspace, folder = _buffered(memory_workspace, {'api.raml': document}) + snapshot = workspace.snapshot(f'{folder}/api.raml') + _title, types, schemes = outline.document_symbols(snapshot, f'{folder}/api.raml') + (facets,) = types.children[0].children + (described,) = schemes.children[0].children + (headers,) = described.children + assert [(each.name, each.selection.line) for each in (facets, described, headers)] == [ + ('facets', _where(document, 'facets:')[0]), + ('describedBy', _where(document, 'describedBy:')[0]), + ('headers', _where(document, 'headers:')[0]), + ] + + def test_an_extension_places_what_it_restated_at_its_own_keys(self, memory_workspace): + # The merge keeps the master's `types:` and `/a:` keys; each document's + # own are recorded before it (docs/21 § 4). + api = '#%RAML 1.0\ntitle: T\ntypes:\n A: string\n/a:\n get:\n' + extension = '#%RAML 1.0 Extension\nextends: api.raml\ntypes:\n B: string\n/a:\n post:\n' + workspace, folder = _buffered(memory_workspace, {'api.raml': api, 'ext.raml': extension}) + found = outline.document_symbols(workspace.snapshot(f'{folder}/ext.raml'), f'{folder}/ext.raml') + assert [(each.name, each.selection.line, each.span.end_line) for each in found] == [ + ('types', _where(extension, 'types:')[0], _where(extension, 'B:')[0]), + ('/a', _where(extension, '/a:')[0], _where(extension, 'post:')[0]), + ] + own = outline.document_symbols(workspace.snapshot(f'{folder}/api.raml'), f'{folder}/api.raml') + assert [(each.name, each.selection.line) for each in own[1:]] == [ + ('types', _where(api, 'types:')[0]), + ('/a', _where(api, '/a:')[0]), + ] def test_a_symbol_spans_its_value_and_selects_its_name(self, parsed): snapshot, folder = parsed From 790cd06c0c1ebabbe911bae59d4f2d3324904bad Mon Sep 17 00:00:00 2001 From: Yury Date: Sun, 11 Oct 2026 01:26:46 +0300 Subject: [PATCH 9/9] docs: record the authored section ranges The dated report holds the scope, the rejected first measurement and the final one; docs/15 and the recovery plan's execution record name the retained-allocation trade-off and that no Stage C candidate remains. Co-Authored-By: Claude Opus 5.5 --- docs/15-implementation-plan.md | 6 +- .../2026-10-11/service-section-ranges.md | 64 +++++++++++++++++++ .../2026-10-10/service-recovery-plan.md | 7 +- 3 files changed, 73 insertions(+), 4 deletions(-) create mode 100644 docs/reports/2026-10-11/service-section-ranges.md diff --git a/docs/15-implementation-plan.md b/docs/15-implementation-plan.md index 55d342c..4a6c76d 100644 --- a/docs/15-implementation-plan.md +++ b/docs/15-implementation-plan.md @@ -33,8 +33,10 @@ but mixed retained allocation rises 7.5%, chiefly the cached outline. An explici acceptance of that new retention trade-off is recorded; the scope, measurements and ownership evidence are in the [shared-index report](reports/2026-10-10/service-shared-indices.md). Cursor-local source lookup replaces hover's whole-file key index -([cursor hover report](reports/2026-10-11/service-cursor-hover.md)). Accurate -authored section ranges remain, measured independently. +([cursor hover report](reports/2026-10-11/service-cursor-hover.md)), and +outline sections are placed at the keys the parser records +([section report](reports/2026-10-11/service-section-ranges.md)); the +recorded sections retain up to 1.65% more on `endpoints`. The record-backed representation replacement remains parked; its recurring rebuild and first-use costs are recorded in the [service cost review](reports/2026-10-10/service-authoring-cost-review.md). diff --git a/docs/reports/2026-10-11/service-section-ranges.md b/docs/reports/2026-10-11/service-section-ranges.md new file mode 100644 index 0000000..bcbc8f0 --- /dev/null +++ b/docs/reports/2026-10-11/service-section-ranges.md @@ -0,0 +1,64 @@ +# Authored section ranges + +Date: 2026-10-11. Branch: `fix/parked-transfers`, measured against its parent +`7e14745`. This is the fourth isolated experiment in the +[recovery plan](../../research/2026-10-10/service-recovery-plan.md). The +section positions are recorded by the parser, so the outline reads only the +model; the alternative, reading the composed source in the outline, was not +taken. + +## 1. Goal and scope + +An outline section (`types:`, a method's `headers:`) spanned its entries and +selected the first, because the model kept no position for its key. So an +empty section was omitted, a `schemas:` table was named `types`, and what an +Extension restated under a master resource was placed at its first entry. + +Decoders now record, in `Raml.written_sections`, each section key an entity +wrote in its own file, with its owner's ID and where its value ends. The +records are held per file; the outline indexes one file's records once. +Recorded: the root's declaration tables, `uses:`, `baseUriParameters:` and +`documentation:`; a type's `facets:`; a resource's `uriParameters:`; a +method's `headers:`, `queryParameters:` and `body:`; a `describedBy:`'s +parameter groups; a response's `headers:` and `body:`; a scheme's +`describedBy:`. + +The merge keeps the master's key where an Overlay or Extension wrote the same +one, so each document's root keys and resource paths are recorded from its +own tree before the merge. A key outside its owner's span, which a template +grafted, is not recorded; nor are the sections of a method or response a +template wrote, since no outline lists them, and recording them would grow +with every application. + +## 2. Correctness + +Tests pin: a section spans its key and entries and selects its key; empty +`traits: {}`, `uriParameters:`, `headers:` and a response's `body:` are +listed; `schemas:` keeps its name; a trait's `queryParameters:` is not the +method's; `facets:`, `describedBy:` and its `headers:` are placed at their +keys; an Extension's `types:` and restated `/a:` are placed at its own keys, +the master's at the master's; a resource type's method and response record +nothing. Over the TCK, every outline entry still holds its selection and lies +in its parent. + +## 3. Results + +Each parse pays one span comparison per section key a decoder meets, and a +record per key kept. On `endpoints`, 4,000 method keys a trait grafted are +rejected and 2,000 response `body:` keys are kept. + +`python -m bench ab 7e14745 --config unwrap`, 5 rounds, `endpoints` again +with 7: + +| workload | time | retained | +|---|---|---| +| `endpoints` | noise (13.7 %) | 17.48 → 17.77 MB (+1.65 %, inside the noise) | +| `templates` | noise | unchanged | +| `large` | noise | +0.3 % | +| `service-navigation` | noise | +0.5 % | +| `service-session` | noise | +0.3 % | + +The first implementation kept a `Position` per span and ignored template +ownership; it cost `endpoints` +7.2 % time and +3.6 % retained, and +`templates` +6.6 % and +5.1 %. Flat per-file records with the end held as two +integers, and skipping template-written owners, removed those. diff --git a/docs/research/2026-10-10/service-recovery-plan.md b/docs/research/2026-10-10/service-recovery-plan.md index 72d4d2d..24b36fa 100644 --- a/docs/research/2026-10-10/service-recovery-plan.md +++ b/docs/research/2026-10-10/service-recovery-plan.md @@ -148,8 +148,11 @@ Current execution: from `069b41f`. Hover reads the keys along the cursor's path instead of a whole-file index; mixed sessions are 9.3% faster and retain 9.6% less, with results in the [cursor hover report](../../reports/2026-10-11/service-cursor-hover.md). - Accurate authored section ranges remain, recorded by the parser so the model - stays self-contained. +- Stage C, fourth bundle: accurate authored section ranges on the same branch, + recorded by the parser so the outline reads only the model. Rebuild time is + within noise; retained allocation rises up to 1.65% on `endpoints`. Results + are in the [section report](../../reports/2026-10-11/service-section-ranges.md). + No Stage C candidate remains. Update these entries as each stage completes. Keep measurements in the dated report, current pending work in docs/15, and execution history here.