diff --git a/bench/ab.py b/bench/ab.py index d236d9c..5bb5ecf 100644 --- a/bench/ab.py +++ b/bench/ab.py @@ -104,15 +104,24 @@ def run_ab( # noqa: PLR0913 - the suite's own selectors, plus the revision and entry = write(name, corpus_root, scale) for config in configs: sides: dict[str, list[Measurement]] = {'A': [], 'B': []} - try: - for _ in range(rounds): - sides['A'].append(spawn(name, config, entry, repeat, cwd=tree)) + unavailable: str | None = None + for _ in range(rounds): + if unavailable is None: + try: + sides['A'].append(spawn(name, config, entry, repeat, cwd=tree)) + except RuntimeError as err: + # A corpus for a feature the revision lacks can + # fail there; that side has no number. + unavailable = str(err) + # B still runs: a broken current worker is a failure, + # not missing evidence. + try: sides['B'].append(spawn(name, config, entry, repeat, cwd=_ROOT)) - except RuntimeError: - # A corpus for a feature the revision lacks can fail - # there; that side has no number, which is the result. - failed = 'A' if len(sides['A']) == len(sides['B']) else 'B' - print(f'{name}/{config:<20} fails in {failed}; not comparable') + except RuntimeError as err: + print(f'{name}/{config} FAIL in B (this tree):\n{err}') + return 1 + if unavailable is not None: + print(f'{name}/{config} fails in A; not comparable (B succeeded):\n{unavailable}') continue print(_row(f'{name}/{config}', sides['A'], sides['B'])) finally: diff --git a/docs/12-performance.md b/docs/12-performance.md index 5de4ab0..82e69a1 100644 --- a/docs/12-performance.md +++ b/docs/12-performance.md @@ -302,7 +302,8 @@ a change to a hot path, or to any code a claim is made about: and this tree over one corpus. A time delta counts only if it is larger than the reported noise. Record the allocation delta whether or not the time moved, because it is nearly deterministic. A field added to every shape - shows there and nowhere else. + shows there and nowhere else. A workload the base revision cannot run is + reported as not comparable; one this tree cannot run fails the command. 3. Where the question is a leaf function's constant factor, add or run a `bench micro` case at sizes that cover the function's range. 4. For a new feature, the base revision has no comparable number. Run @@ -310,18 +311,6 @@ a change to a hot path, or to any code a claim is made about: Include the workload, both deltas, and the noise in the commit message. -A service workload's measured region is a cold snapshot: the parse, plus the -first query of each kind, then reuse of the built indices. Read its delta in -two parts. The parse side is where the base revision built and retained every -file's tree and the comparison does not. The first-use side is where a query -composes on demand the tree the base retained at parse time, one composition -per file per snapshot, about a third of a parse of the file; repeated -queries of the same kind are cached and show no delta. The first-use cost is -avoidable only by retaining the trees, the per-snapshot memory cost -(docs/15 § 2). When a change moves composition from parse time to first use, -name which side the delta is on before attributing it to the change's hot -path (docs/21 § 2, docs/21 § 4). - `bench compare` against the committed baseline is only a coarse check for large regressions: the baseline and the comparison run at different times, and the machine drifts in between. Follow § 5.1 before optimizing and § 5.2 when diff --git a/docs/13-public-api.md b/docs/13-public-api.md index d8e466d..7af0955 100644 --- a/docs/13-public-api.md +++ b/docs/13-public-api.md @@ -136,6 +136,11 @@ Default lint no longer needs whole trees merely to detect `schema:` and retention. A consumer needing only that compact fact should read it rather than retaining the whole source tree. +`Raml.written_sections` holds, per file, each section key an entity wrote +there (`types:`, a method's `headers:`, a type's `facets:`), with its owner's +ID and where its value ends, also independent of source retention. A section +a template contributed is not recorded (docs/21 § 4). + A configuration file's `parser:` section is `ParserConfig`. `limits(options)` applies its `max_include_size`, `max_depth` and `regex_engine`; its `workspace_root` and `remote` are left to the host, which weighs them against diff --git a/docs/15-implementation-plan.md b/docs/15-implementation-plan.md index 308be23..4a6c76d 100644 --- a/docs/15-implementation-plan.md +++ b/docs/15-implementation-plan.md @@ -32,8 +32,11 @@ The shared-index/outline bundle is the next isolated recovery against merged but mixed retained allocation rises 7.5%, chiefly the cached outline. An explicit acceptance of that new retention trade-off is recorded; the scope, measurements and ownership evidence are in the [shared-index report](reports/2026-10-10/service-shared-indices.md). -Remaining candidates are cursor-local source lookup and accurate authored section -ranges, each measured independently. +Cursor-local source lookup replaces hover's whole-file key index +([cursor hover report](reports/2026-10-11/service-cursor-hover.md)), and +outline sections are placed at the keys the parser records +([section report](reports/2026-10-11/service-section-ranges.md)); the +recorded sections retain up to 1.65% more on `endpoints`. The record-backed representation replacement remains parked; its recurring rebuild and first-use costs are recorded in the [service cost review](reports/2026-10-10/service-authoring-cost-review.md). @@ -88,7 +91,8 @@ the media-type fixes and the type-walk work landed together: [integration report](reports/2026-10-10/service-baseline-integration.md). Model-backed declaration inlays remove that first-use work from hints alone; their accepted request-order peak trade-off is recorded in the inlay recovery - report. Cursor-local source lookup remains a separately measured candidate. + report. Hover reads source keys along the cursor's path and keeps no index of + them. ## 3. Potential future work diff --git a/docs/18-linting.md b/docs/18-linting.md index 9b8c333..3013cdd 100644 --- a/docs/18-linting.md +++ b/docs/18-linting.md @@ -226,10 +226,12 @@ fastraml lint [--config FILE] [--severity S] [--ruleset NAME] ``` `--list-rules` and `--explain` require no input document. Other invocations -parse each file with `unwrap=True`, `validate=False`, and `retain_source=True`. -Validation remains the `validate` command's responsibility. A parse failure -produces no findings for that file, reports the parser error, and causes a -nonzero result; remaining files are still attempted. +parse each file with `unwrap=True`, `validate=False` and `retain_text=True`, +which source suppression reads (§ 4); `retain_source=True` only when +`Linter.requires_source` (§ 6). Validation remains the `validate` command's +responsibility. A parse failure produces no findings for that file, reports +the parser error, and causes a nonzero result; remaining files are still +attempted. Formats: diff --git a/docs/21-language-service.md b/docs/21-language-service.md index 33a44cf..89178b8 100644 --- a/docs/21-language-service.md +++ b/docs/21-language-service.md @@ -25,7 +25,7 @@ and several views, and only `cli/` imports it (`docs/02` § 2; | `service/text.py` | converting a column between fastRAML and a protocol | | `service/queries.py` | the queries, in fastRAML positions | | `service/outline.py` | the outline, over the authorship view (`docs/16` § 10) | -| `service/hover.py`, `service/hoverdocs.py` | author-facing hover, source-key indices and explanatory prose (§ 4.2) | +| `service/hover.py`, `service/hoverdocs.py` | author-facing hover, source-key lookup and explanatory prose (§ 4.2) | | `service/datahover.py` | typed `DataNode` key and value spans, shared across value-bearing sites (§ 4.2) | | `service/index.py` | lazy declaration, ID and reverse-hierarchy lookups shared by snapshot queries (§ 4) | | `service/lenses.py` | code-lens sites and on-demand effective RAML type rendering (§ 4.3) | @@ -197,10 +197,18 @@ The first outline request for a URI caches its complete result on the snapshot, including an empty outline. Later requests borrow the same list and symbols; callers must treat them as read-only. Root/dependency edits create a new snapshot and outline cache. A held older snapshot continues to answer from its older model. -Caching does not add authored section positions or populate source grammar. - -The model keeps no position for a section's key (`types:`, a method's -`headers:`), so a section spans its entries and selects the first. +Caching does not populate source grammar. + +A section is placed at the key its owner wrote (`types:`, a method's +`headers:`), spanning the key and its value, and is listed even when empty. +The decoders record these keys in `Raml.written_sections` (docs/13 § 2): the +root's, a type's `facets:`, a resource's `uriParameters:`, a method's or a +`describedBy:`'s `headers:`, `queryParameters:` and `body:`, a response's +`headers:` and `body:`, and a scheme's `describedBy:`. A key is recorded only +inside its owner's span in its file, and not under a method or response a +template wrote, which no outline lists. A table written `schemas:` is named +so. A section the parser recorded no key for spans its entries and selects +the first. What each entry lists is the authorship view's (`docs/16` § 10): a type's own members, not those it inherits, so an inherited property is outlined under the @@ -216,8 +224,10 @@ inside a resource or method (`docs/16` § 10). A file outlines what it wrote, selected by `location` over the model its snapshot parsed. An Extension or Overlay lists the types and other declarations it added to the master's tables, and, under a master resource's -path, the methods and resources it added there: a section spanning them, since -the resource's key it wrote is not in the model. The master lists its own. +path, the methods and resources it added there. The merge keeps the master's +key where both wrote one, so each document's root section keys and resource +paths are recorded from its own tree before the merge: the Extension's +`types:` and restated `/a:` are placed at its keys. The master lists its own. A trait or resource type is listed by name alone. Its body is decoded only where it is applied (`docs/08` § 5), and the model keeps it undecoded, so there @@ -326,14 +336,18 @@ extent is not a fallback for an unknown child. Inline JSON has only the encoded scalar's root span; hover does not invent spans for decoded children. Source-only primitive tokens, including those -in an unapplied template, are read through the type-expression parser and -checked against the source text. No source query binds a reference. Explanatory +in an unapplied template, are read through the type-expression parser. A +token's offset is its column only where the scalar's one-line span is its +text, or its text in quotes; where an escape or a tag shifts the columns, no +token is found. No source query binds a reference. Explanatory prose is a documentation catalogue, not a table that accepts fields or overrides parser diagnostics. -Hover indices and formatted subjects are lazy per snapshot. Source keys are indexed once per queried -file from retained nodes, or a composition of its retained text when source -trees were not retained. These nodes and indices die with the snapshot. +Hover indices and formatted subjects are lazy per snapshot. Source keys are not +indexed: each hover reads the path to its cursor (`syntax.keys_at`), one entry +per mapping level found by binary search, in the file's retained nodes, or a +composition of its retained text when source trees were not retained. Those +nodes die with the snapshot. The typed-data token index is owned by the snapshot, lazy and built once; formatted data targets are cached by hover. The same typed-data targets supply go-to-definition for nested field keys and scalar values. A definition request may populate that @@ -501,12 +515,6 @@ value, so an integer larger than a double reaches a JavaScript client as written. It is the preview's source in `contrib/fastraml-vscode` (`docs/17` § 4). -**Latency.** On `large`, an edit costs 475 ms and allocates 48.8 MB before -its parser diagnostics, against 366 ms for a plain `unwrap+validate` parse -(`python -m bench run --bench large --config service`). About a quarter of -the parse composes the unchanged libraries: the most a compose cache (G8) -could save. - ## 6. Verification - `test_service_text.py`: conversion in each encoding, both ways. @@ -518,12 +526,14 @@ could save. lifetime, retained original trees and distinct JSON-include normalization. - `test_service_queries.py`: each query on one document with a library, a DataType include, a trait and a resource type; every query on a parse - stopped at each stage; an Extension's outline; and, over the TCK, that every - outline entry holds its selection and lies in its parent. + stopped at each stage; an Extension's outline and its own section keys; + empty, `schemas:` and template-supplied sections; and, over the TCK, that + every outline entry holds its selection and lies in its parent. - `test_service_hover.py`: contextual field meanings, full Markdown prose, authored and inherited summaries, aliases, optional versus nullable values, source-only template help, token boundaries, opaque data, educational examples, - and user-defined facet descriptions at declarations and supplied keys. + and user-defined facet descriptions at declarations and supplied keys; over + the TCK, the cursor path reaches every key the grammar walk yields. - `test_service_data_hover.py`: shared nested-field and scalar-value help for custom facets, annotations, examples, defaults and enums; array items, inheritance, patterns, discriminated and ambiguous unions, recursive types, diff --git a/docs/README.md b/docs/README.md index 7cff42b..9f1ec21 100644 --- a/docs/README.md +++ b/docs/README.md @@ -67,6 +67,9 @@ architecture. The implementation and these documents must agree. for explaining past decisions. It is not normative. - `research/` contains open investigations. Nothing there defines parser behavior or may be required by the implementation. +- `reports/YYYY-MM-DD/` contains dated benchmark measurements, audits and other + point-in-time records. Results stay in those reports; the numbered documents + describe current contracts and behavior, not individual runs. Current parser work is complete. Deferred work is listed in the non-normative [status and roadmap](15-implementation-plan.md). Conformance status belongs in diff --git a/docs/reports/2026-10-11/service-cursor-hover.md b/docs/reports/2026-10-11/service-cursor-hover.md new file mode 100644 index 0000000..c506d2c --- /dev/null +++ b/docs/reports/2026-10-11/service-cursor-hover.md @@ -0,0 +1,54 @@ +# Cursor-local hover source lookup + +Date: 2026-10-11. Branch: `fix/parked-transfers`, from master `069b41f`; +measured against its parent `131b08e`. This is the third isolated experiment +in the [recovery plan](../../research/2026-10-10/service-recovery-plan.md). + +## 1. Goal, workload and cache boundaries + +The goal is to remove the whole-file source-key index from the first hover in +a file, without slowing repeated hovers or changing which token is explained. + +Before: the first hover in a URI sorted every key `syntax.keys` yields and +parsed every type value for built-in tokens, checked against a split of the +file's text. Both lists lived as long as the snapshot: 15,203 keys and about +2.6 MB on the hover corpus (one 13,204-line file). Later hovers bisected them. + +After: each hover reads `syntax.keys_at`, the keys enclosing the cursor, one +per mapping level, found by binary search over that mapping's entries, which +are in document order. Sequences are scanned, because an alias item keeps its +anchor's earlier position. Only the type value the cursor is in is parsed for +built-in tokens. Nothing is cached; the composed tree is still the snapshot's +or the shared source owner's (docs/21 § 4), so the cost of composition is +unchanged. Include-context discovery for headerless files is unchanged. + +A token's offset is taken as its column where the scalar's one-line span is +its text, or its text in quotes, rather than by comparing a line of the source. +An escape or a tag lengthens the span, so those tokens stay unexplained, as +before. A type expression folded over several lines now gets no token help; +before, a token on its first line could. + +## 2. Correctness and reach + +Over every TCK `.raml` file, `keys_at` at each key's start returns that key +with the same site and table as the whole-file walk (more than 10,000 keys). +A quoted union explains `nil` at its exact span; an escaped `n\x69l` explains +nothing. The existing hover suite, including unapplied-template built-ins, +opaque data, expanded aliases and inherited include contexts, passes +unchanged. The inlay reach test now fails if `Hover._keys_at` is called. + +## 3. Results + +`python -m bench ab 131b08e --bench hover --bench service-session --config unwrap --rounds 5`: + +| workload | time | noise | peak | retained | +|---|---|---|---|---| +| `hover` | 332.9 → 337.3 ms, noise | 16.2 % | unchanged | 26.29 → 23.71 MB (−9.8 %) | +| `service-session` | 1260.5 → 1142.8 ms (−9.3 %) | 7.4 % | unchanged | 27.50 → 24.87 MB (−9.6 %) | + +`hover` runs one probe per declaration, so its time is dominated by repeated +hovers; the removed index was paid once. `service-session` hovers sparsely +after each of three edits, so it pays the index once per edit and gains. + +`python -m bench linearity --bench hover`: time 0.992, peak 0.992, +retained 0.998. diff --git a/docs/reports/2026-10-11/service-section-ranges.md b/docs/reports/2026-10-11/service-section-ranges.md new file mode 100644 index 0000000..bcbc8f0 --- /dev/null +++ b/docs/reports/2026-10-11/service-section-ranges.md @@ -0,0 +1,64 @@ +# Authored section ranges + +Date: 2026-10-11. Branch: `fix/parked-transfers`, measured against its parent +`7e14745`. This is the fourth isolated experiment in the +[recovery plan](../../research/2026-10-10/service-recovery-plan.md). The +section positions are recorded by the parser, so the outline reads only the +model; the alternative, reading the composed source in the outline, was not +taken. + +## 1. Goal and scope + +An outline section (`types:`, a method's `headers:`) spanned its entries and +selected the first, because the model kept no position for its key. So an +empty section was omitted, a `schemas:` table was named `types`, and what an +Extension restated under a master resource was placed at its first entry. + +Decoders now record, in `Raml.written_sections`, each section key an entity +wrote in its own file, with its owner's ID and where its value ends. The +records are held per file; the outline indexes one file's records once. +Recorded: the root's declaration tables, `uses:`, `baseUriParameters:` and +`documentation:`; a type's `facets:`; a resource's `uriParameters:`; a +method's `headers:`, `queryParameters:` and `body:`; a `describedBy:`'s +parameter groups; a response's `headers:` and `body:`; a scheme's +`describedBy:`. + +The merge keeps the master's key where an Overlay or Extension wrote the same +one, so each document's root keys and resource paths are recorded from its +own tree before the merge. A key outside its owner's span, which a template +grafted, is not recorded; nor are the sections of a method or response a +template wrote, since no outline lists them, and recording them would grow +with every application. + +## 2. Correctness + +Tests pin: a section spans its key and entries and selects its key; empty +`traits: {}`, `uriParameters:`, `headers:` and a response's `body:` are +listed; `schemas:` keeps its name; a trait's `queryParameters:` is not the +method's; `facets:`, `describedBy:` and its `headers:` are placed at their +keys; an Extension's `types:` and restated `/a:` are placed at its own keys, +the master's at the master's; a resource type's method and response record +nothing. Over the TCK, every outline entry still holds its selection and lies +in its parent. + +## 3. Results + +Each parse pays one span comparison per section key a decoder meets, and a +record per key kept. On `endpoints`, 4,000 method keys a trait grafted are +rejected and 2,000 response `body:` keys are kept. + +`python -m bench ab 7e14745 --config unwrap`, 5 rounds, `endpoints` again +with 7: + +| workload | time | retained | +|---|---|---| +| `endpoints` | noise (13.7 %) | 17.48 → 17.77 MB (+1.65 %, inside the noise) | +| `templates` | noise | unchanged | +| `large` | noise | +0.3 % | +| `service-navigation` | noise | +0.5 % | +| `service-session` | noise | +0.3 % | + +The first implementation kept a `Position` per span and ignored template +ownership; it cost `endpoints` +7.2 % time and +3.6 % retained, and +`templates` +6.6 % and +5.1 %. Flat per-file records with the end held as two +integers, and skipping template-written owners, removed those. diff --git a/docs/research/2026-10-10/service-recovery-plan.md b/docs/research/2026-10-10/service-recovery-plan.md index 5f9532c..24b36fa 100644 --- a/docs/research/2026-10-10/service-recovery-plan.md +++ b/docs/research/2026-10-10/service-recovery-plan.md @@ -143,8 +143,16 @@ Current execution: The implemented bundle passes local correctness/reach/scaling checks and improves sparse navigation 17.0%; its 7.5% mixed retained-allocation increase is attributed mainly to the requested-file outline cache. The full bundle is accepted with - that retention trade-off, subject to the normal CI/platform gate. - Remaining candidates stay separate experiments against the integrated baseline. + that retention trade-off. All CI checks passed; PR #4 merged at `069b41f`. +- Stage C, third bundle: cursor-local source lookup on `fix/parked-transfers` + from `069b41f`. Hover reads the keys along the cursor's path instead of a + whole-file index; mixed sessions are 9.3% faster and retain 9.6% less, with + results in the [cursor hover report](../../reports/2026-10-11/service-cursor-hover.md). +- Stage C, fourth bundle: accurate authored section ranges on the same branch, + recorded by the parser so the outline reads only the model. Rebuild time is + within noise; retained allocation rises up to 1.65% on `endpoints`. Results + are in the [section report](../../reports/2026-10-11/service-section-ranges.md). + No Stage C candidate remains. Update these entries as each stage completes. Keep measurements in the dated report, current pending work in docs/15, and execution history here. diff --git a/fastraml/cli/lint.py b/fastraml/cli/lint.py index 07504d4..57d85f9 100644 --- a/fastraml/cli/lint.py +++ b/fastraml/cli/lint.py @@ -73,7 +73,9 @@ def _lint(args: argparse.Namespace) -> int: # noqa: PLR0911, PLR0912 - command findings = [] failed = False for path in args.files: - raml = parse_or_report(args, path, retain_source=True) + # The text serves suppression directives (docs/18 § 4); the trees only + # a source-sensitive rule. + raml = parse_or_report(args, path, retain_text=True, retain_source=linter.requires_source) if raml is None: failed = True continue diff --git a/fastraml/parser/extensions.py b/fastraml/parser/extensions.py index 9d4142f..0a20b40 100644 --- a/fastraml/parser/extensions.py +++ b/fastraml/parser/extensions.py @@ -34,6 +34,7 @@ ReferenceResolver, identify_fragment, load_fragment_text, + record_root_sections, resolve_uses, unmarshal_uses, ) @@ -82,6 +83,8 @@ def decode_extension_chain(raml: Raml, uri: str, kind: FragmentKind, text: str) accumulator.add(RamlError.new('title is required', root_api.uri, API_HEAD_SPAN)) target = root_api.root + # Each document's own: the merge keeps the master's key where two wrote one. + record_root_sections(raml, api, target) declared_by: dict[str, dict[str, int]] = {} fragments: list[ExtensionFragment] = [] for position, document in enumerate(documents, start=1): @@ -92,6 +95,8 @@ def decode_extension_chain(raml: Raml, uri: str, kind: FragmentKind, text: str) fragment.api = api fragments.append(_register(raml, fragment)) _decode_own_keys(raml, fragment, document, accumulator) + record_root_sections(raml, fragment, document.root) + _record_resources(raml, fragment, document.root, '') result = merge_extension( target, document.root, @@ -220,6 +225,19 @@ def _decode_own_keys(raml: Raml, fragment: ExtensionFragment, document: _Documen raml.pop_ctx() +def _record_resources(raml: Raml, fragment: ExtensionFragment, node: Node, path: str) -> None: + """The resource keys an Overlay or Extension wrote, named by full path: one + it restates is merged into the master's, whose key the model keeps. + """ + if node.kind is not NodeKind.MAPPING: + return + for key, value in pairs(node): + if key.value.startswith('/'): + full = path + key.value + raml.record_section(fragment, key, value, fragment.location, name=full) + _record_resources(raml, fragment, value, full) + + def _resolve_libraries( raml: Raml, api: APIFragment, fragments: list[ExtensionFragment], accumulator: Accumulator ) -> None: diff --git a/fastraml/parser/fragments.py b/fastraml/parser/fragments.py index 389dad1..711dad6 100644 --- a/fastraml/parser/fragments.py +++ b/fastraml/parser/fragments.py @@ -125,6 +125,7 @@ 'parse_fragment', 'parse_included_fragment', 'parse_library', + 'record_root_sections', 'resolve_uses', ] @@ -1276,6 +1277,36 @@ def parse_fragment(raml: Raml, uri: str, kind: FragmentKind) -> Fragment: return decode_fragment(raml, uri, authored_kind, text) +#: The root keys that open a section of declarations or settings (docs/21 § 4). +_ROOT_SECTIONS: Final = frozenset( + { + FACET_USES, + FACET_TYPES, + FACET_SCHEMAS, + FACET_ANNOTATION_TYPES, + FACET_TRAITS, + FACET_RESOURCE_TYPES, + FACET_SECURITY_SCHEMES, + FACET_BASE_URI_PARAMETERS, + FACET_DOCUMENTATION, + } +) +_DOCUMENT_KINDS: Final = frozenset( + {FragmentKind.API, FragmentKind.LIBRARY, FragmentKind.OVERLAY, FragmentKind.EXTENSION} +) + + +def record_root_sections(raml: Raml, fragment: Fragment, root: Node) -> None: + """The section keys `fragment`'s own document wrote at its root, before + any merge: a fragment that is one declaration has `uses:` only. + """ + if root.kind is not NodeKind.MAPPING: + return + for key, value in pairs(root): + if key.value == FACET_USES or (key.value in _ROOT_SECTIONS and fragment.kind in _DOCUMENT_KINDS): + raml.record_section(fragment, key, value, fragment.location) + + def decode_fragment(raml: Raml, uri: str, kind: FragmentKind, text: str) -> Fragment: """Register, decode, then resolve `uses:` — in that order. See the module docstring.""" raml.store_source_text(uri, text) @@ -1302,6 +1333,7 @@ def decode_fragment(raml: Raml, uri: str, kind: FragmentKind, text: str) -> Frag try: root = compose(text, uri=uri, max_depth=raml.max_depth, key_pool=raml.mapping_keys) raml.store_source_node(uri, root) + record_root_sections(raml, fragment, root) fragment.decode(root) except RamlError as err: # The `uses:` that decoded are still resolved, so a mistake in the body diff --git a/fastraml/parser/security.py b/fastraml/parser/security.py index d7baec8..22ed312 100644 --- a/fastraml/parser/security.py +++ b/fastraml/parser/security.py @@ -29,6 +29,8 @@ FACET_DESCRIBED_BY, FACET_DESCRIPTION, FACET_DISPLAY_NAME, + FACET_HEADERS, + FACET_QUERY_PARAMETERS, FACET_REQUEST_TOKEN_URI, FACET_RESPONSES, FACET_SCOPES, @@ -223,6 +225,7 @@ def make_security_scheme_definition( # noqa: PLR0912 - one pass over the declar elif name == FACET_DESCRIPTION: definition.description = make_string_facet(raml, key, value, location) elif name == FACET_DESCRIBED_BY: + raml.record_section(definition, key, value, location) _decode_described_by(raml, value, location, partial(setattr, definition, 'described_by')) elif name == FACET_SETTINGS: settings_node = value @@ -263,11 +266,13 @@ def _decode_described_by( accumulator = Accumulator() for key, value in pairs(node): name = key.value + if name in (FACET_HEADERS, FACET_QUERY_PARAMETERS): + raml.record_section(description, key, value, location) try: if decode_request_facet(raml, description, key, value, location): continue if name == FACET_RESPONSES: - decode_responses(raml, value, location, description.responses) + decode_responses(raml, value, location, description.responses, holder=description) elif is_annotation_key(name): add_domain_extension(raml, description.annotations, location, key, value) else: diff --git a/fastraml/parser/source_decode.py b/fastraml/parser/source_decode.py index 9560569..d22e9d2 100644 --- a/fastraml/parser/source_decode.py +++ b/fastraml/parser/source_decode.py @@ -45,6 +45,7 @@ from fastraml.parser.facets import MEDIA_RANGE, make_string_facet from fastraml.parser.includes import inline_include from fastraml.parser.syntax import is_media_type_map as _is_media_type_map +from fastraml.registry import written_inside from fastraml.types.shape import make_body_shape, make_parameter_map, make_shape from fastraml.yamlnode import NodeKind, is_null, node_error, pairs @@ -57,7 +58,13 @@ from fastraml.types.base import BaseShape, Parameter from fastraml.yamlnode import Node -__all__ = ['decode_request_facet', 'decode_responses', 'decode_source_endpoint', 'query_exclusion_error'] +__all__ = [ + 'REQUEST_SECTIONS', + 'decode_request_facet', + 'decode_responses', + 'decode_source_endpoint', + 'query_exclusion_error', +] @contextmanager @@ -181,12 +188,16 @@ def _is_status_code(value: str) -> bool: return _STATUS_CODE.match(value) is not None -def _decode_response(raml: Raml, key: Node, value: Node, location: str, attach: Callable[[Response], None]) -> None: +def _decode_response( # noqa: PLR0913 - the response's pair, where it goes, and its holder + raml: Raml, key: Node, value: Node, location: str, attach: Callable[[Response], None], *, holder: object | None +) -> None: """One response, attached before its content is decoded. A response whose content fails stays attached, marked in `Raml.broken` - (docs/13 § 1). + (docs/13 § 1). Its section keys are recorded only where `holder` wrote + it, not a template. """ + written = holder is not None and written_inside(holder, key) location = raml.location_of(value, location) response = Response( id=raml.next_id(), @@ -208,6 +219,8 @@ def _decode_response(raml: Raml, key: Node, value: Node, location: str, attach: with raml.target_scope(DomainLocation.RESPONSE): for child_key, child_value in pairs(value): name = child_key.value + if written and name in (FACET_HEADERS, FACET_BODY): + raml.record_section(response, child_key, child_value, location) try: if name == FACET_DISPLAY_NAME: response.display_name = make_string_facet(raml, child_key, child_value, location) @@ -228,9 +241,12 @@ def _decode_response(raml: Raml, key: Node, value: Node, location: str, attach: accumulator.raise_if_any() -def decode_responses(raml: Raml, node: Node, location: str, responses: dict[str, Response]) -> None: +def decode_responses( + raml: Raml, node: Node, location: str, responses: dict[str, Response], *, holder: object | None +) -> None: """A `responses:` map, into the holder's own. Public because `describedBy:` - reuses it verbatim. + reuses it verbatim. `holder` is the entity that wrote it, or `None` for + one a template wrote, whose section keys are not recorded. """ node, location = inline_include(raml, node, location) if is_null(node): @@ -243,7 +259,7 @@ def decode_responses(raml: Raml, node: Node, location: str, responses: dict[str, try: if not _is_status_code(key.value): raise node_error('status code must be a 3-digit number', location, key, info={'code': key.value}) - _decode_response(raml, key, value, location, partial(setitem, responses, key.value)) + _decode_response(raml, key, value, location, partial(setitem, responses, key.value), holder=holder) except RamlError as err: accumulator.add(err) accumulator.raise_if_any() @@ -260,6 +276,10 @@ class RequestFacets(Protocol): query_string: BaseShape | None +#: The keys of a method that open a section (docs/21 § 4). +REQUEST_SECTIONS: Final = frozenset({FACET_HEADERS, FACET_QUERY_PARAMETERS, FACET_BODY}) + + def decode_request_facet(raml: Raml, into: RequestFacets, key: Node, value: Node, location: str) -> bool: """`headers`, `queryParameters` or `queryString`. Returns whether it was one.""" name = key.value @@ -290,8 +310,10 @@ def query_exclusion_error(raml: Raml, facets: RequestFacets, location: str, node def _decode_operation_field( # noqa: PLR0913, PLR0917 - one pass over the method's key vocabulary - raml: Raml, operation: Operation, request: Request, key: Node, value: Node, location: str + raml: Raml, operation: Operation, request: Request, key: Node, value: Node, location: str, *, written: bool ) -> None: + if written and key.value in REQUEST_SECTIONS: + raml.record_section(operation, key, value, location) if decode_request_facet(raml, request, key, value, location): return name = key.value @@ -304,18 +326,22 @@ def _decode_operation_field( # noqa: PLR0913, PLR0917 - one pass over the metho elif name == FACET_BODY: _decode_bodies(raml, key, value, location, DomainLocation.REQUEST_BODY, request.bodies) elif name == FACET_RESPONSES: - decode_responses(raml, value, location, operation.responses) + decode_responses(raml, value, location, operation.responses, holder=operation if written else None) elif is_annotation_key(name): add_domain_extension(raml, operation.annotations, location, key, value) else: raml.recover(node_error('unknown field', location, key, info={'field': name})) -def decode_source_operation(raml: Raml, source: SourceOperation, attach: Callable[[Operation], None]) -> None: +def decode_source_operation( + raml: Raml, source: SourceOperation, attach: Callable[[Operation], None], *, holder: EndPoint +) -> None: """One method's retained tree into an `Operation`, attached before its content is decoded. One whose content fails stays attached, - marked in `Raml.broken` (docs/13 § 1). + marked in `Raml.broken` (docs/13 § 1). Its section keys are recorded only + where `holder` wrote it, not a resource type. """ + written = written_inside(holder, source.key_pos) operation = Operation( id=raml.next_id(), method=source.method, @@ -349,7 +375,9 @@ def decode_source_operation(raml: Raml, source: SourceOperation, attach: Callabl # survive the containers the merge synthesised. A pair a # template grafted is located where the template was written. with raml.provenance_scope(value): - _decode_operation_field(raml, operation, request, key, value, raml.location_of(key, location)) + _decode_operation_field( + raml, operation, request, key, value, raml.location_of(key, location), written=written + ) except RamlError as err: accumulator.add(err) accumulator.add(query_exclusion_error(raml, request, location, source.body)) @@ -368,6 +396,7 @@ def _decode_endpoint_field(raml: Raml, endpoint: EndPoint, key: Node, value: Nod elif name == FACET_DESCRIPTION: endpoint.description = make_string_facet(raml, key, value, location) elif name == FACET_URI_PARAMETERS: + raml.record_section(endpoint, key, value, location) make_parameter_map(raml, value, location, 'uri', endpoint.uri_parameters) elif is_annotation_key(name): add_domain_extension(raml, endpoint.annotations, location, key, value) @@ -420,7 +449,9 @@ def decode_source_endpoint(raml: Raml, source: SourceEndPoint, attach: Callable[ for method, operation_source in source.operations.items(): try: - decode_source_operation(raml, operation_source, partial(setitem, endpoint.operations, method)) + decode_source_operation( + raml, operation_source, partial(setitem, endpoint.operations, method), holder=endpoint + ) except RamlError as err: accumulator.add(err) diff --git a/fastraml/parser/syntax.py b/fastraml/parser/syntax.py index 11bf789..0dc73bb 100644 --- a/fastraml/parser/syntax.py +++ b/fastraml/parser/syntax.py @@ -2,12 +2,14 @@ These are mapping positions, not annotation targets or a second resolver. `child_site` is the extension merge's existing grammar (docs/19 § 3.1). -`keys` projects a composed source tree into those positions for an editor -(docs/21 § 4.2); data and application arguments stay opaque. +`keys` projects a composed source tree into those positions for an editor, +and `keys_at` the path to one cursor (docs/21 § 4.2); data and application +arguments stay opaque. """ from __future__ import annotations +from bisect import bisect_right from dataclasses import dataclass from enum import Enum, auto from typing import TYPE_CHECKING, Final @@ -32,6 +34,7 @@ 'fragment_site', 'is_media_type_map', 'keys', + 'keys_at', ] #: Resource methods; optional `?` spellings belong to resource-type templates. @@ -197,6 +200,51 @@ def annotation(self) -> bool: return self.site not in NAME_MAPS and is_annotation_key(self.node.value) +_OPAQUE: Final = frozenset({Site.DATA, Site.APPLICATION, Site.GENERIC}) + + +def keys_at(root: Node, site: Site, line: int, column: int, *, table: str = '') -> list[Key]: + """The keys `keys` yields whose entry encloses `line:column`, outermost first. + + Each level follows the last entry written at or before the position, so + one lookup reads one path rather than the file. The last key holds the + position, or the value it lies in, or neither when it lies past the end. + """ + position = (line, column) + found: list[Key] = [] + node = root + while site not in _OPAQUE: + if node.kind is NodeKind.SEQUENCE: + # Linear: an alias item keeps its anchor's earlier position. + start = (node.line, node.column) + items = [item for item in node.content if start <= (item.line, item.column) <= position] + if not items: + break + node = items[-1] + continue + if node.kind is not NodeKind.MAPPING: + break + if site is Site.BODY and not is_media_type_map(node): + if any('/' in key.value for key, _ in pairs(node)): + break + site = Site.TYPE + content = node.content + # A mapping's own entries are in document order. + index = bisect_right( + range(len(content) // 2), position, key=lambda i: (content[2 * i].line, content[2 * i].column) + ) + if not index: + break + key, value = content[2 * index - 2], content[2 * index - 1] + found.append(Key(key, value, node, site, table)) + if key.position.holds(line, column) or (value.line, value.column) < (key.line, key.column): + break + child = child_site(site, key.value) + table = key.value if child in NAME_MAPS else table + node, site = value, child + return found + + def keys(root: Node, site: Site, *, table: str = '') -> Iterator[Key]: """Source keys, without decoding data or expanding templates. @@ -206,7 +254,7 @@ def keys(root: Node, site: Site, *, table: str = '') -> Iterator[Key]: stack = [(root, site, table)] while stack: node, context, table = stack.pop() - if context in (Site.DATA, Site.APPLICATION, Site.GENERIC): + if context in _OPAQUE: continue if node.kind is NodeKind.SEQUENCE: stack.extend( diff --git a/fastraml/registry.py b/fastraml/registry.py index d02bb72..a63dc3b 100644 --- a/fastraml/registry.py +++ b/fastraml/registry.py @@ -23,8 +23,9 @@ from fastraml.domains import DomainLocation from fastraml.errors import Accumulator, RamlError from fastraml.loaders import SchemeLoader -from fastraml.sourceinfo import KeywordUse -from fastraml.yamlnode import AUTHORED_NODES, DEFAULT_MAX_DEPTH, mark_subtree +from fastraml.positions import UNKNOWN, Position +from fastraml.sourceinfo import KeywordUse, WrittenSection +from fastraml.yamlnode import AUTHORED_NODES, DEFAULT_MAX_DEPTH, mark_subtree, written_end if TYPE_CHECKING: from collections.abc import Iterator, Mapping, Sequence @@ -36,7 +37,6 @@ from fastraml.parser.includes import IncludeRef from fastraml.parser.structural_merge import ProvenanceOverlay from fastraml.parser.substitutions import Substitutions - from fastraml.positions import Position from fastraml.types.base import Property from fastraml.types.expressions import ExprCache from fastraml.types.schema_compile import SchemaRegistry @@ -79,6 +79,24 @@ class Stage(Enum): VALIDATED = 'validated' # P10, when requested +def written_inside(owner: object, node: Node | Position) -> bool: + """Whether `node` starts inside `owner`'s span, from its key through its + value; true for an owner placed nowhere. Compared as numbers: a parse + asks once per section key. + """ + key: Position = getattr(owner, 'key_pos', UNKNOWN) + value: Position = getattr(owner, 'value_pos', UNKNOWN) + if not key.is_known: + key = value + if not value.is_known: + value = key + if not key.is_known: + return True + start = min((key.line, key.column), (value.line, value.column)) + end = max((key.end_line, key.end_column), (value.end_line, value.end_column)) + return start <= (node.line, node.column) <= end + + class Identified(Protocol): """Anything `Raml.broken` can mark: every model entity has an id.""" @@ -213,6 +231,7 @@ class Raml: 'shapes', 'substitutions', 'syntax_aliases', + 'written_sections', # --- work queues ----------------------------------------------------- '_discriminator_shapes', 'unresolved_shapes', @@ -302,6 +321,9 @@ def __init__( # noqa: PLR0913 - the parse's configuration, keyword-only, one fi #: Accepted compatibility spellings by entity ID; only used spellings #: allocate a record. The parser records syntax, a view judges it. self.syntax_aliases: dict[int, list[KeywordUse]] = {} + #: The section keys entities wrote, by the file they wrote them in, in + #: the order decoded (docs/21 § 4). + self.written_sections: dict[str, list[WrittenSection]] = {} # A worklist, drained from the left in P7 while resolution appends to # the right; a deque keeps both ends O(1). @@ -644,6 +666,21 @@ def record_syntax_alias(self, entity: Identified, key: Node, location: str) -> N use = KeywordUse(self.location_of(key, location), key.position, key.value) self.syntax_aliases.setdefault(entity.id, []).append(use) + def record_section(self, owner: Identified, key: Node, value: Node, location: str, *, name: str = '') -> None: + """Where `owner` wrote a section key (docs/21 § 4). + + Only in `owner`'s file and, where it is placed, inside its span: a + section a template grafted was written in the template. `name`, when + given, replaces the key's: a resource's full path. + """ + if not written_inside(owner, key): + return + where = self.location_of(key, location) + if where != getattr(owner, 'location', where): + return + section = WrittenSection(owner.id, name or key.value, key.position, *written_end(value)) + self.written_sections.setdefault(where, []).append(section) + def put_source_info(self, entity_id: int, key: Node | None, value: Node) -> None: """Index an entity's authored nodes when source retention is on.""" if self.source_info is not None: diff --git a/fastraml/service/hover.py b/fastraml/service/hover.py index d724120..31db2b3 100644 --- a/fastraml/service/hover.py +++ b/fastraml/service/hover.py @@ -1,9 +1,10 @@ """Author-facing hover: source explanations and compact model summaries. The service composes the parser's source grammar, occurrences and authorship -view. It binds no name and runs no pass (docs/21 § 4.2). Indices and source -keys are built once per snapshot, so successive hovers do not rescan the whole -model. Formatted subjects and declaration hints are cached. +view. It binds no name and runs no pass (docs/21 § 4.2). Indices are built +once per snapshot, so successive hovers do not rescan the whole model; source +keys are read along the cursor's path only. Formatted subjects and declaration +hints are cached. """ from __future__ import annotations @@ -19,7 +20,7 @@ from fastraml.parser.fragments import APIFragment, LibraryLink from fastraml.parser.includes import is_json_ref from fastraml.parser.security import SecuritySchemeDefinition -from fastraml.parser.syntax import METHODS, NAME_MAPS, Key, Site, child_site, fragment_site, keys +from fastraml.parser.syntax import METHODS, NAME_MAPS, Key, Site, child_site, fragment_site, keys, keys_at from fastraml.parser.templates import TemplateDefinition from fastraml.service.datahover import DataHover, DataTarget, data_roots from fastraml.service.hoverdocs import BUILTINS, METHOD_DOCS, field_doc @@ -34,7 +35,6 @@ underlying_type, ) from fastraml.service.source import original_tree -from fastraml.service.text import Lines from fastraml.types.base import BaseShape, Parameter, facets_of from fastraml.types.complex_ import ArrayShape, ObjectShape, UnionShape, UnknownShape from fastraml.types.expressions import Array, Optional_, Primitive, Union, parse_expression @@ -117,8 +117,6 @@ class Hover: '_descriptions', '_includes', '_inlay_declarations', - '_key_starts', - '_keys', '_mappings', '_nodes', '_occurrences', @@ -130,7 +128,6 @@ class Hover: '_semantic', '_sites', '_source_backend', - '_source_builtins', '_source_generation', '_sources', '_subjects', @@ -159,12 +156,9 @@ def __init__( # noqa: PLR0913 - the snapshot's borrowed source owner and compos self._by_uri: dict[str, list[_Subject]] = {} self._sites: dict[tuple[str, int, int], _Subject] = {} self._mappings: dict[tuple[str, int, int], _Subject] = {} - self._keys: dict[str, list[Key]] = {} - self._key_starts: dict[str, list[tuple[int, int]]] = {} self._resources: dict[tuple[str, int, int], EndPoint] = {} self._operations: dict[tuple[str, int, int], tuple[str, Operation]] = {} self._responses: dict[tuple[str, int, int], Response] = {} - self._source_builtins: dict[str, Occurrences] = {} self._ambiguous_sites: set[tuple[str, int, int]] = set() self._nodes: dict[str, Node | None] = {} self._contexts: dict[str, set[tuple[Site, str]]] = {} @@ -281,7 +275,8 @@ def _bodies(self, bodies: Iterable[Body], role: str) -> None: def at(self, uri: str, line: int, column: int) -> tuple[str, Position] | None: """Explain the precise token under the cursor, or return no answer.""" - key = self._key_at(uri, line, column) + path = self._keys_at(uri, line, column) + key = path[-1] if path and path[-1].node.position.holds(line, column) else None if key is not None: text = self._key_doc(uri, key) if text is not None: @@ -299,9 +294,9 @@ def at(self, uri: str, line: int, column: int) -> tuple[str, Position] | None: data = self._data.at(uri, line, column) if data: return self._data_docs(data), data[0].span - primitive = self._source_builtins[uri].at(uri, line, column) - if primitive: - return self._builtin(primitive[0].written), primitive[0].span + primitive = self._builtin_at(path[-1], line, column) if path and key is None else None + if primitive is not None: + return self._builtin(primitive[0]), primitive[1] if key is not None: subject = self._sites.get((uri, key.node.line, key.node.column)) text = self._describe(subject) if subject is not None else _unbound_doc(key) @@ -475,59 +470,42 @@ def _discover_contexts(self) -> None: contexts.add(context) pending.append((target, child, child_table)) - def _source_keys(self, uri: str) -> None: + def _keys_at(self, uri: str, line: int, column: int) -> list[Key]: + """The source keys enclosing the cursor, read along its path only.""" if uri not in self._contexts and not self._contexts_ready: self._discover_contexts() contexts = self._contexts.get(uri, {(Site.DATA, '')}) context, table = next(iter(contexts)) if len(contexts) == 1 else (Site.DATA, '') if context in (Site.DATA, Site.APPLICATION, Site.GENERIC): - self._keys[uri] = [] - self._key_starts[uri] = [] - self._source_builtins[uri] = Occurrences((), ()) - return + return [] node = self._node(uri) - found = ( - [] - if node is None - else sorted(keys(node, context, table=table), key=lambda key: (key.node.line, key.node.column)) - ) - self._keys[uri] = found - self._key_starts[uri] = [(key.node.line, key.node.column) for key in found] - lines = Lines(self._raml.source_texts.get(uri, '')) - primitives = [] - cache: ExprCache = {} - for key in found: - value = key.type_value() - if value is None: - continue - if value.tag != TAG_STR: - continue - tokens = [] - if value.value in BUILTINS: - tokens.append((value.value, 0)) - elif _OPERATORS.search(value.value): - with suppress(RamlError): - cached = self._raml.expr_cache.get(value.value) - if cached is not None: - cache[value.value] = cached - tree = parse_expression(value.value, cache) - tokens.extend((token.name, token.col) for token in _primitives(tree)) - for name, offset in tokens: - span = value.position.within(value.value).shifted(offset, len(name)) - if lines.line(span.line)[span.column - 1 : span.end_column - 1] == name: - primitives.append( - Occurrence(uri, span.line, span.column, span.end_column, Role.BUILTIN, Kind.TYPE, None, name) - ) - self._source_builtins[uri] = Occurrences(primitives, ()) - - def _key_at(self, uri: str, line: int, column: int) -> Key | None: - if uri not in self._keys: - self._source_keys(uri) - index = bisect_right(self._key_starts[uri], (line, column)) - if index: - found = self._keys[uri][index - 1] - if found.node.position.holds(line, column): - return found + return [] if node is None else keys_at(node, context, line, column, table=table) + + def _builtin_at(self, key: Key, line: int, column: int) -> tuple[str, Position] | None: + """A built-in type named by the type expression the cursor is in.""" + value = key.type_value() + if value is None or value.tag != TAG_STR or value.line != line or value.end_line != line: + return None + text = value.value + # Token offsets map onto columns only where the span is the text, or + # the text in quotes: an escape or a tag shifts them. + if value.end_column - value.column not in (len(text), len(text) + 2): + return None + tokens: list[tuple[str, int]] = [] + if text in BUILTINS: + tokens.append((text, 0)) + elif _OPERATORS.search(text): + with suppress(RamlError): + cache: ExprCache = {} + cached = self._raml.expr_cache.get(text) + if cached is not None: + cache[text] = cached + tokens.extend((token.name, token.col) for token in _primitives(parse_expression(text, cache))) + start = value.position.within(text) + for name, offset in tokens: + span = start.shifted(offset, len(name)) + if span.column <= column < span.end_column: + return name, span return None def _key_doc(self, uri: str, key: Key) -> str | None: diff --git a/fastraml/service/outline.py b/fastraml/service/outline.py index 1aa708c..9b79a34 100644 --- a/fastraml/service/outline.py +++ b/fastraml/service/outline.py @@ -24,12 +24,33 @@ from fastraml.parser.documentation import DocumentationItem from fastraml.parser.endpoints import Body, Operation, Request, Response from fastraml.parser.templates import TemplateDefinition + from fastraml.registry import Identified, Raml from fastraml.service.workspace import Snapshot from fastraml.types.base import Parameter, ScalarFacet __all__ = ['document_symbols'] +class _Written: + """The section keys entities wrote in one file (`Raml.written_sections`), + by owner, indexed once per outline. + """ + + __slots__ = ('_by_owner',) + + def __init__(self, raml: Raml, uri: str) -> None: + self._by_owner: dict[int, dict[str, _Placed]] = {} + for each in raml.written_sections.get(uri, ()): + self._by_owner.setdefault(each.owner, {}).setdefault(each.name, (uri, each.span, each.key)) + + def of(self, owner: Identified) -> Mapping[str, _Placed]: + return self._by_owner.get(owner.id, {}) + + +#: Where a section key was written: its file, its span and the key's own span. +type _Placed = tuple[str, Position, Position] + + def document_symbols(snapshot: Snapshot, uri: str) -> list[Symbol]: """Borrow the read-only outline cached for this URI and snapshot.""" if uri not in snapshot.outlines: @@ -43,8 +64,8 @@ def _document_symbols(snapshot: Snapshot, uri: str) -> list[Symbol]: Declarations only, as a code outline lists them: a type and its members, not its examples or annotations. A type's detail is its type as written, - and an optional member is named `name?`. A section the model records no - key for, `types:` or a method's `headers:`, spans its entries. + and an optional member is named `name?`. A section is placed at the key + its owner wrote, which the parser records, and listed even when empty. What an Extension or an Overlay adds is outlined in it: a type it declared in the master's table, and under a master resource's path, a method it @@ -54,30 +75,37 @@ def _document_symbols(snapshot: Snapshot, uri: str) -> list[Symbol]: semantic = snapshot.semantic if raml is None or semantic is None or uri not in raml.fragments: return [] + written = _Written(raml, uri) + root = written.of(raml.fragments[uri]) found: list[Symbol | None] = [ symbol(name, SymbolKind.METADATA, facet, facet.value) for name, facet in authored.metadata(raml, uri) ] - parameters = (_parameter(name, param) for name, param in authored.base_uri_parameters(raml, uri)) - found.append(_group('baseUriParameters', parameters)) - found.append(_group('documentation', map(_documentation, authored.documentation(raml, uri)))) + parameters = (_parameter(name, param, written) for name, param in authored.base_uri_parameters(raml, uri)) + found.append(_group('baseUriParameters', parameters, root)) + found.append(_group('documentation', map(_documentation, authored.documentation(raml, uri)), root)) uses = (symbol(name, SymbolKind.LIBRARY, link, link.value) for name, link in authored.uses(raml, uri).items()) - found.append(_group('uses', uses)) - sections: dict[str, list[Symbol | None]] = {} + found.append(_group('uses', uses, root)) + sections: dict[str, list[Symbol | None]] = {key: [] for key in DECLARATION_KINDS if _spelled(key, root) in root} for key, name, entity in semantic.by_uri.get(uri, ()): - sections.setdefault(key, []).append(_declaration(name, DECLARATION_KINDS[key], entity)) - found += (_group(key, entries) for key, entries in sections.items()) - found += (_resource(written) for written in authored.resources(raml, uri)) - found += _fragment_body(authored.fragment_body(raml, uri)) + sections.setdefault(key, []).append(_declaration(name, DECLARATION_KINDS[key], entity, written)) + found += (_group(_spelled(key, root), entries, root) for key, entries in sections.items()) + found += (_resource(each, written, root) for each in authored.resources(raml, uri)) + found += _fragment_body(authored.fragment_body(raml, uri), written) return _here(uri, found) -def _fragment_body(body: BaseShape | SecuritySchemeDefinition | None) -> list[Symbol | None]: +def _spelled(key: str, written: Mapping[str, _Placed]) -> str: + """The table's key as the file wrote it: `schemas` is `types` in the model.""" + return 'schemas' if key == 'types' and 'schemas' in written and 'types' not in written else key + + +def _fragment_body(body: BaseShape | SecuritySchemeDefinition | None, written: _Written) -> list[Symbol | None]: """What a fragment file that is one declaration wrote, at the top: the file is the declaration, and its name is the file's. """ if isinstance(body, BaseShape): - return _member_symbols(body) - return [] if body is None else [_described(body)] + return _member_symbols(body, written) + return [] if body is None else [_described(body, written)] def _documentation(item: DocumentationItem) -> Symbol | None: @@ -86,61 +114,61 @@ def _documentation(item: DocumentationItem) -> Symbol | None: return None if title is None else symbol(str(title.value), SymbolKind.DOCUMENTATION, item, key=title.value_pos) -def _declaration(name: str, kind: SymbolKind, entity: object) -> Symbol | None: +def _declaration(name: str, kind: SymbolKind, entity: object, written: _Written) -> Symbol | None: if isinstance(entity, BaseShape): - return _type(name, kind, entity) + return _type(name, kind, entity, written) if isinstance(entity, SecuritySchemeDefinition): found = symbol(name, kind, entity, entity.resolved().type) if found is not None: - _adopt(found, [_described(entity)]) + _adopt(found, [_described(entity, written)]) return found return symbol(name, kind, cast('TemplateDefinition', entity)) -def _type(name: str, kind: SymbolKind, base: BaseShape) -> Symbol | None: +def _type(name: str, kind: SymbolKind, base: BaseShape, written: _Written) -> Symbol | None: """A type-like symbol: `base` and the members it declares.""" found = symbol(name, kind, base, _written(base)) if found is None: return None found.form = 'enum' if base.enum is not None else base.type or '' - _members(base, found) + _adopt(found, _member_symbols(base, written)) return found -def _described(definition: SecuritySchemeDefinition) -> Symbol | None: +def _described(definition: SecuritySchemeDefinition, written: _Written) -> Symbol | None: """A security scheme's `describedBy`, as the definition wrote it.""" described = definition.described_by - return None if described is None else _group('describedBy', _message(described, definition)) - - -def _members(base: BaseShape, parent: Symbol) -> None: - """Add to `parent` the members `base` declares.""" - _adopt(parent, _member_symbols(base)) + if described is None: + return None + children = _message(described, definition, written, written.of(described)) + return _group('describedBy', children, written.of(definition)) -def _member_symbols(base: BaseShape) -> list[Symbol | None]: +def _member_symbols(base: BaseShape, written: _Written) -> list[Symbol | None]: """The members `base` declares: not those it inherits, nor `items` an expression built (docs/16 § 10). """ found: list[Symbol | None] = [ - _type(key if prop.required else f'{key}?', SymbolKind.PROPERTY, prop.base) + _type(key if prop.required else f'{key}?', SymbolKind.PROPERTY, prop.base, written) for key, prop in authored.properties(base) ] found += ( - _type(f'/{key}/', SymbolKind.PROPERTY, pattern.base) for key, pattern in authored.pattern_properties(base) + _type(f'/{key}/', SymbolKind.PROPERTY, pattern.base, written) + for key, pattern in authored.pattern_properties(base) ) items = authored.items(base) if items is not None: - found.append(_type('items', SymbolKind.TYPE, items)) + found.append(_type('items', SymbolKind.TYPE, items, written)) facets = ( - _type(key if prop.required else f'{key}?', SymbolKind.FACET, prop.base) for key, prop in authored.facets(base) + _type(key if prop.required else f'{key}?', SymbolKind.FACET, prop.base, written) + for key, prop in authored.facets(base) ) - found.append(_group('facets', facets)) + found.append(_group('facets', facets, written.of(base))) return found -def _parameter(name: str, param: Parameter) -> Symbol | None: - return _type(name if param.required else f'{name}?', SymbolKind.PARAMETER, param.declaration.base) +def _parameter(name: str, param: Parameter, written: _Written) -> Symbol | None: + return _type(name if param.required else f'{name}?', SymbolKind.PARAMETER, param.declaration.base, written) def _written(base: BaseShape) -> str: @@ -156,77 +184,92 @@ def _written(base: BaseShape) -> str: return written.value.lstrip() -def _resource(written: authored.WrittenResource) -> Symbol | None: +def _resource(resource: authored.WrittenResource, written: _Written, root: Mapping[str, _Placed]) -> Symbol | None: """A resource as this file wrote it: at its key when the file declares - it, or else a section, named by its path, holding what the file added. + it, or else at the key it restated it under, holding what it added. """ - endpoint = written.endpoint - children: list[Symbol | None] = [_method(operation) for operation in written.operations] - children += (_resource(child) for child in written.resources) - if not written.here: - return _group(endpoint.uri, children, SymbolKind.RESOURCE) + endpoint = resource.endpoint + children: list[Symbol | None] = [_method(operation, written) for operation in resource.operations] + children += (_resource(child, written, root) for child in resource.resources) + if not resource.here: + # An Overlay or Extension recorded the path it restated (docs/21 § 4). + return _group(endpoint.uri, children, root, SymbolKind.RESOURCE, key=endpoint.full_uri) found = symbol(endpoint.uri, SymbolKind.RESOURCE, endpoint, _shown(endpoint.display_name)) if found is None: return None applied = () if endpoint.resource_type is None else (endpoint.resource_type,) - parameters = (_parameter(name, param) for name, param in authored.parameters(endpoint, endpoint.uri_parameters)) + parameters = ( + _parameter(name, param, written) for name, param in authored.parameters(endpoint, endpoint.uri_parameters) + ) _adopt( found, [ _applied('type', authored.members(endpoint, applied)), _applied('is', authored.members(endpoint, endpoint.traits)), _applied('securedBy', authored.secured_by(endpoint)), - _group('uriParameters', parameters), + _group('uriParameters', parameters, written.of(endpoint)), *children, ], ) return found -def _method(operation: Operation) -> Symbol | None: +def _method(operation: Operation, written: _Written) -> Symbol | None: found = symbol(operation.method, SymbolKind.METHOD, operation, _shown(operation.display_name)) if found is None: return None + sections = written.of(operation) children: list[Symbol | None] = [ _applied('is', authored.members(operation, operation.traits)), _applied('securedBy', authored.secured_by(operation)), ] if operation.request is not None: - children += _message(operation.request, operation) - children.append(_group('body', _bodies(operation.request.bodies, operation))) - children += (_response(response) for response in authored.members(operation, operation.responses.values())) + children += _message(operation.request, operation, written, sections) + children.append(_group('body', _bodies(operation.request.bodies, operation, written), sections)) + children += (_response(response, written) for response in authored.members(operation, operation.responses.values())) _adopt(found, children) return found def _message( - message: Request | SecuritySchemeDescription, owner: Operation | SecuritySchemeDefinition + message: Request | SecuritySchemeDescription, + owner: Operation | SecuritySchemeDefinition, + written: _Written, + sections: Mapping[str, _Placed], ) -> list[Symbol | None]: """A request's, or a `describedBy`'s, parameter groups and query string, - as `owner` wrote them; a `describedBy`'s responses too. + as `owner` wrote them, with the section keys it recorded; a `describedBy`'s + responses too. """ - query = (_parameter(name, param) for name, param in authored.parameters(owner, message.query_parameters)) - headers = (_parameter(name, param) for name, param in authored.parameters(owner, message.headers)) - found: list[Symbol | None] = [_group('queryParameters', query), _group('headers', headers)] + query = (_parameter(name, param, written) for name, param in authored.parameters(owner, message.query_parameters)) + headers = (_parameter(name, param, written) for name, param in authored.parameters(owner, message.headers)) + found: list[Symbol | None] = [_group('queryParameters', query, sections), _group('headers', headers, sections)] string = message.query_string if string is not None and authored.wrote(owner, string.location, string.key_pos): - found.append(_type('queryString', SymbolKind.TYPE, string)) + found.append(_type('queryString', SymbolKind.TYPE, string, written)) if isinstance(message, SecuritySchemeDescription): - found += (_response(response) for response in authored.members(owner, message.responses.values())) + found += (_response(response, written) for response in authored.members(owner, message.responses.values())) return found -def _response(response: Response) -> Symbol | None: +def _response(response: Response, written: _Written) -> Symbol | None: shown = _shown(response.display_name) or _shown(response.description) found = symbol(response.code, SymbolKind.RESPONSE, response, shown) if found is None: return None - headers = (_parameter(name, param) for name, param in authored.parameters(response, response.headers)) - _adopt(found, [_group('headers', headers), _group('body', _bodies(response.bodies, response))]) + sections = written.of(response) + headers = (_parameter(name, param, written) for name, param in authored.parameters(response, response.headers)) + _adopt( + found, + [ + _group('headers', headers, sections), + _group('body', _bodies(response.bodies, response, written), sections), + ], + ) return found -def _bodies(bodies: Mapping[str, Body], owner: Operation | Response) -> Iterator[Symbol | None]: +def _bodies(bodies: Mapping[str, Body], owner: Operation | Response, written: _Written) -> Iterator[Symbol | None]: """One symbol per body `owner` wrote: a `body:` with no media type is one body per default media type (docs/08 § 6.3), named by all of them. """ @@ -236,7 +279,7 @@ def _bodies(bodies: Mapping[str, Body], owner: Operation | Response) -> Iterator # The body's own key: its shape's is none for `body: Book`. found = symbol(name, SymbolKind.BODY, body, '' if shape is None else _written(shape)) if found is not None and shape is not None: - _members(shape, found) + _adopt(found, _member_symbols(shape, written)) yield found @@ -254,12 +297,24 @@ def _applied(name: str, refs: Iterable[DirectiveRef | SecurityScheme]) -> Symbol ) -def _group(name: str, children: Iterable[Symbol | None], kind: SymbolKind = SymbolKind.SECTION) -> Symbol | None: - """A section holding `children`, in the order written, or `None` for an - empty one. The model keeps no position for a section's key, so it spans - its entries and selects the first. +def _group( + name: str, + children: Iterable[Symbol | None], + sections: Mapping[str, _Placed], + kind: SymbolKind = SymbolKind.SECTION, + *, + key: str = '', +) -> Symbol | None: + """A section holding `children`, in the order written, at the key its + owner wrote, `sections[key or name]`. One the parser recorded no key for, + such as a section a template supplied, spans its entries and selects the + first, or is `None` when empty. """ placed = _ordered(children) + at = sections.get(key or name) + if at is not None: + uri, span, selection = at + return Symbol(name, kind, uri, span, selection, placed) if not placed: return None first = placed[0] diff --git a/fastraml/service/workspace.py b/fastraml/service/workspace.py index d4c6d88..87a5f74 100644 --- a/fastraml/service/workspace.py +++ b/fastraml/service/workspace.py @@ -438,6 +438,12 @@ def _parse(self, root: str) -> Snapshot: else RamlError.wrap('load resource', err, root, kind=ErrorKind.READING) ) return Snapshot(root, None, failure, frozenset({root})) + texts = raml.source_texts + for uri, kept in texts.items(): + # The parse decoded its own copy of an open buffer's text; keep the + # buffer's instead of a second one for the snapshot's lifetime. + if (buffer := self.buffers.get(uri)) is not None and buffer.text == kept: + texts[uri] = buffer.text return Snapshot( root, raml, diff --git a/fastraml/skilldata/lint/SKILL.md b/fastraml/skilldata/lint/SKILL.md index edb267e..1a8afe3 100644 --- a/fastraml/skilldata/lint/SKILL.md +++ b/fastraml/skilldata/lint/SKILL.md @@ -313,7 +313,7 @@ invalid document to trigger, it is the wrong rule. - **`lint needs an unwrapped model`** — you are calling the Python API directly. Parse with `ParseOptions(unwrap=True)`. - **`enabled lint rules need retained source`** — the same, with a rule enabled - that reads the source text. Also pass `retain_source=True`. + that reads the YAML trees. Also pass `retain_source=True`. - **A plugin's rules never run** — discovery is not activation. Name the plugin under `plugins:` in the config. - **`duplicate rule id`** — two rules claim one name and registration refuses diff --git a/fastraml/sourceinfo.py b/fastraml/sourceinfo.py index 2e566be..5e5dca3 100644 --- a/fastraml/sourceinfo.py +++ b/fastraml/sourceinfo.py @@ -1,17 +1,16 @@ -"""Facts a decoder records when it accepts a compatibility spelling. +"""Facts a decoder records about how a document was written. -A `KeywordUse` names where a deprecated keyword was written. It is a model -fact the registry stores and a view may report on: not a diagnostic, and it -keeps no YAML node (docs/04 § 5, docs/05 § 3). +A `KeywordUse` names where a deprecated keyword was written; a `WrittenSection` +where an entity wrote a section key, such as `types:` or a method's `headers:`. +Each is a model fact the registry stores and a view may read: not a +diagnostic, and it keeps no YAML node (docs/04 § 5, docs/05 § 3, docs/21 § 4). """ from __future__ import annotations from dataclasses import dataclass -from typing import TYPE_CHECKING -if TYPE_CHECKING: - from fastraml.positions import Position +from fastraml.positions import Position @dataclass(frozen=True, slots=True, eq=False) @@ -21,3 +20,24 @@ class KeywordUse: location: str position: Position name: str + + +@dataclass(frozen=True, slots=True, eq=False) +class WrittenSection: + """A section key one entity wrote in its own file, and where the entry + it opens ends. Held per file, so the record carries its owner's ID. + + `name` is the key as written (`schemas`, not `types`), or for a resource + an Extension restated, the resource's full path. + """ + + owner: int + name: str + key: Position + end_line: int + end_column: int + + @property + def span(self) -> Position: + """From the key through the end of its value.""" + return Position(self.key.line, self.key.column, self.end_line, self.end_column) diff --git a/fastraml/types/shape.py b/fastraml/types/shape.py index 773caa2..9ddbda8 100644 --- a/fastraml/types/shape.py +++ b/fastraml/types/shape.py @@ -387,6 +387,7 @@ def _decode( # noqa: PLR0912 - one pass over the common-facet vocabulary (docs/ case fn.FACET_REQUIRED: base.required = make_bool_facet(raml, key, value, location) case fn.FACET_FACETS: + raml.record_section(base, key, value, location) _decode_custom_facet_defs(raml, base, value) case fn.FACET_EXAMPLE: _decode_example(raml, base, key, value) diff --git a/fastraml/views/authored.py b/fastraml/views/authored.py index 5e28fd5..a4a1a10 100644 --- a/fastraml/views/authored.py +++ b/fastraml/views/authored.py @@ -125,10 +125,15 @@ def members[P: Placed](owner: Placed, found: Iterable[P]) -> Iterator[P]: def parameters(owner: Placed, written: Mapping[str, Parameter]) -> Iterator[tuple[str, Parameter]]: """The parameters `owner` wrote. A parameter is a record placed at its - key, in the file its shape was written in. + key, in the file its shape was written in; one synthesized for an + undeclared URI variable is placed at its resource's key and written by no one. """ test = _writer(owner) - return ((name, param) for name, param in written.items() if test(param.base.location, param.key_pos)) + return ( + (name, param) + for name, param in written.items() + if not param.synthesized and test(param.base.location, param.key_pos) + ) def secured_by(owner: EndPoint | Operation) -> Iterator[SecurityScheme]: diff --git a/fastraml/views/lint/rules/style.py b/fastraml/views/lint/rules/style.py index 0378d00..2395223 100644 --- a/fastraml/views/lint/rules/style.py +++ b/fastraml/views/lint/rules/style.py @@ -354,12 +354,14 @@ def type_(self, ctx: Context, iri: str, base: BaseShape, shape_kind: str) -> Ite def api(self, ctx: Context, iri: str, api: APIFragment) -> Iterable[Finding]: if api.description is not None: return () - root = ctx.raml.source_node(api.location) - position = UNKNOWN if root is None else root.position + # The API has no key of its own: it is placed at the `title` in force, + # which an Overlay or Extension may have written, and read from the + # model, so no rule in the default set needs the trees (docs/18 § 6). + title = api.title + location = api.location if title is None else title.location + position = UNKNOWN if title is None else title.key_pos return ( - ctx.at( - self.meta, 'entity has no description', location=api.location, position=position, iri=iri, entity='API' - ), + ctx.at(self.meta, 'entity has no description', location=location, position=position, iri=iri, entity='API'), ) def endpoint(self, ctx: Context, iri: str, endpoint: EndPoint) -> Iterable[Finding]: diff --git a/fastraml/yamlnode.py b/fastraml/yamlnode.py index 7108f48..357df32 100644 --- a/fastraml/yamlnode.py +++ b/fastraml/yamlnode.py @@ -50,6 +50,7 @@ 'with_content', 'with_grafts', 'with_value', + 'written_end', ] try: # pragma: no cover - depends on how PyYAML was built @@ -490,6 +491,15 @@ def _end(node: Node) -> tuple[int, int]: return end +def written_end(node: Node) -> tuple[int, int]: + """Where `full_position` ends, without building the `Position`.""" + if not node.content: + return node.end_line, node.end_column + if isinstance(node, _Grafted): + return node.written_end + return _end(node) + + def end_line(node: Node) -> int: """The last line `full_position` spans.""" return node.full_position.end_line diff --git a/tests/bench/test_ab.py b/tests/bench/test_ab.py new file mode 100644 index 0000000..29363a0 --- /dev/null +++ b/tests/bench/test_ab.py @@ -0,0 +1,59 @@ +"""`bench ab`: a revision lacking a feature is not a broken current tree (docs/12 § 5).""" + +from __future__ import annotations + +import pytest + +from bench import ab +from bench.harness import Measurement + + +@pytest.mark.parametrize('fail_a', [None, 1, 2]) +@pytest.mark.parametrize('fail_b', [None, 1, 2]) +def test_a_failing_base_is_not_comparable_and_a_failing_current_tree_fails( + tmp_path, monkeypatch, capsys, fail_a, fail_b +): + base, current = tmp_path / 'base', tmp_path / 'current' + monkeypatch.setattr(ab, '_ROOT', current) + monkeypatch.setattr(ab, '_checkout', lambda ref: base) + monkeypatch.setattr(ab, '_imports_from', lambda tree: tree / 'fastraml' / '__init__.py') + git_calls = [] + + def git(*args): + git_calls.append(args) + return 'base' + + monkeypatch.setattr(ab, '_git', git) + counts = {'A': 0, 'B': 0} + roots = [] + + def write(name, root, scale): + roots.append(root) + return root / 'api.raml' + + def spawn(name, config, entry, repeat, *, cwd): + side = 'A' if cwd == base else 'B' + counts[side] += 1 + if counts[side] == (fail_a if side == 'A' else fail_b): + raise RuntimeError(f'{side} worker stderr') + return Measurement(name, config, seconds=1, allocated_bytes=100, max_rss_bytes=None, retained_bytes=50) + + result = ab.run_ab('base', ['feature'], ['unwrap'], scale=0.5, repeat=2, rounds=3, spawn=spawn, write=write) + output = capsys.readouterr().out + assert result == (0 if fail_b is None else 1) + assert output.count('feature/unwrap') == 1 + assert counts['B'] == (3 if fail_b is None else fail_b) + if fail_b is not None: + assert 'feature/unwrap FAIL in B (this tree)' in output + assert 'B worker stderr' in output + elif fail_a is not None: + assert 'feature/unwrap fails in A; not comparable (B succeeded)' in output + assert 'A worker stderr' in output + assert counts['A'] == fail_a + else: + assert 'noise' in output + assert 'not comparable' not in output + # The base worktree and the corpus are removed however the run ends. + assert git_calls[-1] == ('worktree', 'remove', '--force', str(base)) + assert roots + assert not any(root.exists() for root in roots) diff --git a/tests/bench/test_corpus.py b/tests/bench/test_corpus.py index 4d09bb0..5d5fbc4 100644 --- a/tests/bench/test_corpus.py +++ b/tests/bench/test_corpus.py @@ -180,7 +180,7 @@ def test_inlays_reaches_inferred_types_data_types_and_inherited_facets(self, tmp original = inlays.inlay_hints def source(*args, **kwargs): - pytest.fail('model-backed inlays must not compose source or populate grammar/builtin indices') + pytest.fail('model-backed inlays must not compose source or read its grammar') def hints(snapshot, uri, span): nonlocal passes @@ -190,7 +190,7 @@ def hints(snapshot, uri, span): return result monkeypatch.setattr(Hover, '_node', source) - monkeypatch.setattr(Hover, '_source_keys', source) + monkeypatch.setattr(Hover, '_keys_at', source) monkeypatch.setattr(inlays, 'inlay_hints', hints) entry = corpus.write_hover(tmp_path, family_count=count) run_one('inlays', 'unwrap', entry, repeat=1) diff --git a/tests/unit/test_lint.py b/tests/unit/test_lint.py index 951e287..eb59bb9 100644 --- a/tests/unit/test_lint.py +++ b/tests/unit/test_lint.py @@ -773,10 +773,39 @@ def api(self, ctx, iri, api): 'unknown-position' ] - def test_api_finding_can_be_suppressed_at_the_root_mapping(self, tmp_path): - source = '#%RAML 1.0\n# fastraml: ignore missing-description\ntitle: t\n' + @pytest.mark.parametrize('retention', ['retain_text', 'retain_source']) + @pytest.mark.parametrize('suppressed', [False, True]) + def test_api_finding_is_placed_and_suppressed_at_its_title_without_trees(self, workspace, retention, suppressed): + directive = '# fastraml: ignore missing-description\n' if suppressed else '' + source = '#%RAML 1.0\nversion: v1\n' + directive + 'title: t\n' + linter = Linter(builtin_registry(), Config(extends=(), rules=(RuleSetting(id='missing-description'),))) + assert not linter.requires_source + raml = workspace.document(source, ParseOptions(unwrap=True, **{retention: True})) + expected = [] if suppressed else [({'entity': 'API'}, raml.location, Position(3, 1, 3, 6))] + assert [(finding.info, finding.location, finding.position) for finding in linter.run(raml)] == expected + + @pytest.mark.parametrize('kind', ['Overlay', 'Extension']) + @pytest.mark.parametrize('replaced', [False, True]) + def test_api_finding_follows_the_title_in_force(self, workspace, kind, replaced): + root = workspace( + { + 'api.raml': '#%RAML 1.0\nversion: v1\ntitle: Original\n', + 'changed.raml': f'#%RAML 1.0 {kind}\nextends: api.raml\nusage: Changed\n' + + ('title: Changed\n' if replaced else ''), + } + ) + raml = workspace.parse(root / 'changed.raml', ParseOptions(unwrap=True, retain_text=True)) config = Config(extends=(), rules=(RuleSetting(id='missing-description'),)) - assert not Linter(builtin_registry(), config).run(parsed(source, tmp_path)) + findings = Linter(builtin_registry(), config).run(raml) + place = ((root / 'changed.raml').as_uri(), 4) if replaced else ((root / 'api.raml').as_uri(), 3) + assert [(finding.location, finding.position.line) for finding in findings] == [place] + + def test_api_finding_without_a_title_has_no_invented_position(self, workspace): + raml = workspace.document('#%RAML 1.0\ntitle: T\n', ParseOptions(unwrap=True, retain_source=True)) + raml.entry_point.title = None + config = Config(extends=(), rules=(RuleSetting(id='missing-description'),)) + findings = Linter(builtin_registry(), config).run(raml) + assert [(finding.location, finding.position.is_known) for finding in findings] == [(raml.location, False)] def test_suppression_uses_an_included_files_location(self, workspace): root = workspace( @@ -1492,6 +1521,30 @@ def test_default_warnings_do_not_fail_the_run(self, disk_workspace, capsys): assert 'deprecated-schemas' in output assert 'WARN 0 errors, 1 warning and 1 info finding.' in output + @pytest.mark.parametrize( + ('rules', 'trees'), [([], False), (['prefer-array-expression'], True), (['prefer-array-expression=off'], False)] + ) + def test_trees_are_kept_only_for_a_source_sensitive_rule(self, disk_workspace, capsys, monkeypatch, rules, trees): + # docs/18 § 5: the text always, for the suppression the directive asks. + from fastraml.cli import lint + + source = '#%RAML 1.0\ntitle: T\n# fastraml: ignore deprecated-schemas\nschemas:\n T: string\n' + root = disk_workspace({'api.raml': source}) + models = [] + original = lint.parse_or_report + + def capture(*args, **kwargs): + models.append(original(*args, **kwargs)) + return models[-1] + + monkeypatch.setattr(lint, 'parse_or_report', capture) + arguments = ['lint', str(root / 'api.raml')] + for rule in rules: + arguments.extend(['--rule', rule]) + assert main(arguments) == EXIT_OK + assert [(raml.retain_text, raml.retain_source) for raml in models] == [(True, trees)] + assert 'deprecated-schemas' not in capsys.readouterr().out + def test_human_color_is_tty_only_and_can_be_disabled(self, disk_workspace, capsys, monkeypatch): root = disk_workspace({'api.raml': '#%RAML 1.0\ntitle: t\nschemas:\n U: string\n'}) monkeypatch.delenv('NO_COLOR', raising=False) diff --git a/tests/unit/test_service_hover.py b/tests/unit/test_service_hover.py index 809ee12..9ee8e1d 100644 --- a/tests/unit/test_service_hover.py +++ b/tests/unit/test_service_hover.py @@ -282,6 +282,18 @@ def test_a_builtin_explains_allowed_values_and_supported_facets(self, hover): assert '`pattern`' in string assert 'RAML reference' in integer + def test_a_builtin_is_found_in_quotes_but_not_where_an_escape_shifts_its_columns(self, memory_workspace): + # docs/21 § 4.2: a token's offset in the expression is its column only + # where the span is the text, or the text in quotes. + document = '#%RAML 1.0\ntitle: T\ntypes:\n A:\n type: "string | nil"\n B:\n type: "n\\x69l"\n' + workspace, folder = _buffered(memory_workspace, {'api.raml': document}) + snapshot = workspace.snapshot(f'{folder}/api.raml') + assert snapshot.error is None + text, span = queries.hover(snapshot, f'{folder}/api.raml', *_where(document, 'nil')) + assert 'Accepts only null' in text + assert (span.column, span.end_column) == (_where(document, 'nil')[1], _where(document, 'nil')[1] + 3) + assert queries.hover(snapshot, f'{folder}/api.raml', *_where(document, 'x69')) is None + def test_hover_does_not_extend_past_the_target_token(self, memory_workspace): document = '#%RAML 1.0\ntitle: T\ntypes:\n Data:\n properties: {} # comment\n' workspace, folder = _buffered(memory_workspace, {'api.raml': document}) @@ -496,3 +508,33 @@ def test_method_help_explains_http_behavior(self, hover): assert 'Retrieves a representation' in text assert 'without requesting a change to server state' in text assert 'possible responses' in text + + +@pytest.mark.tck +def test_the_cursor_path_finds_every_key_the_whole_file_walk_yields(): + # docs/21 § 4.2: hover reads the path to one cursor, which must reach every + # key the grammar walk projects, in the same position and table. + from fastraml.errors import RamlError + from fastraml.parser.syntax import fragment_site, keys, keys_at + from fastraml.yamlnode import compose + from tests.tck.conftest import tck_root + + root = tck_root() + if root is None: + pytest.skip('no TCK corpus; set FASTRAML_TCK_DIR') + checked, missed = 0, [] + for path in sorted(root.rglob('*.raml')): + text = path.read_text(encoding='utf-8-sig', errors='replace') + header = text.partition('\n')[0].removeprefix('#%RAML 1.0').strip() + try: + node = compose(text, uri=path.as_uri()) + except RamlError: + continue + site = fragment_site(header or 'API') + for key in keys(node, site): + found = keys_at(node, site, key.node.line, key.node.column) + checked += 1 + if not found or (found[-1].node, found[-1].site, found[-1].table) != (key.node, key.site, key.table): + missed.append(f'{path.relative_to(root).as_posix()}:{key.node.line}:{key.node.column}') + assert checked > 10_000 + assert not missed, missed[:20] diff --git a/tests/unit/test_service_queries.py b/tests/unit/test_service_queries.py index 11e0bd8..4ff224b 100644 --- a/tests/unit/test_service_queries.py +++ b/tests/unit/test_service_queries.py @@ -277,6 +277,22 @@ def test_the_outline_holds_what_each_declaration_writes_and_not_what_it_inherits ]), ] # fmt: skip + def test_an_undeclared_uri_variable_is_not_outlined(self, memory_workspace): + # docs/08 § 6.2: P6 synthesizes a `string` for `{part}` and `{x}`, placed + # at the resource's key, which the author wrote; the parameter they did not. + document = '#%RAML 1.0\ntitle: T\n/items/{id}/{part}:\n uriParameters:\n id: string\n get:\n/a/{x}:\n' + workspace, folder = _buffered(memory_workspace, {'api.raml': document}) + snapshot = workspace.snapshot(f'{folder}/api.raml') + section, resource = SymbolKind.SECTION, SymbolKind.RESOURCE + assert _tree(outline.document_symbols(snapshot, f'{folder}/api.raml')) == [ + ('title', SymbolKind.METADATA, 'T'), + ('/items/{id}/{part}', resource, '', [ + ('uriParameters', section, '', [('id', SymbolKind.PARAMETER, 'string')]), + ('get', SymbolKind.METHOD, ''), + ]), + ('/a/{x}', resource, ''), + ] # fmt: skip + def test_a_redeclared_pattern_is_outlined_where_it_was_written(self, memory_workspace): """docs/07 § 4 puts a redeclared `/b/` at P's place after unwrap; the authored view and the outline follow the author, who wrote `/c/` first. @@ -322,7 +338,8 @@ def outlined(name: str): ('401', SymbolKind.RESPONSE, ''), ])] # fmt: skip assert outlined('home.raml') == [('documentation', section, '', [('Home', SymbolKind.DOCUMENTATION, '')])] - assert [each[0] for each in outlined('api.raml')] == ['title', 'securitySchemes', 'types'] + # `documentation:` is written here, its one item in its own file. + assert [each[0] for each in outlined('api.raml')] == ['title', 'documentation', 'securitySchemes', 'types'] def test_a_documentation_item_file_no_api_reads_outlines_its_title(self, memory_workspace): files = {'home.raml': '#%RAML 1.0 DocumentationItem\ntitle: Home\ncontent: hi\n'} @@ -371,12 +388,101 @@ def test_a_file_outlines_what_it_wrote_an_extension_what_it_added(self, memory_w ('/a', resource, '', [('get', method, '')]), ] - def test_a_section_spans_its_entries_and_selects_the_first(self, parsed): + def test_a_section_spans_its_key_and_entries_and_selects_its_key(self, parsed): + # docs/21 § 4: the decoder records the key the owner wrote. snapshot, folder = parsed types = next(s for s in outline.document_symbols(snapshot, f'{folder}/api.raml') if s.name == 'types') - entity, book = types.children - assert (types.span.line, types.span.end_line) == (entity.span.line, book.span.end_line) - assert types.selection == entity.selection + _entity, book = types.children + line, column = _where(API, 'types:') + assert (types.span.line, types.span.column, types.span.end_line) == (line, column, book.span.end_line) + assert (types.selection.line, types.selection.column, types.selection.end_column) == (line, column, column + 5) + + def test_an_empty_or_deprecated_section_is_outlined_at_its_key(self, memory_workspace): + document = ( + '#%RAML 1.0\ntitle: T\nschemas:\n A: string\ntraits: {}\n' + '/r:\n uriParameters:\n get:\n headers:\n responses:\n 200:\n body:\n' + ) + workspace, folder = _buffered(memory_workspace, {'api.raml': document}) + snapshot = workspace.snapshot(f'{folder}/api.raml') + found = outline.document_symbols(snapshot, f'{folder}/api.raml') + section = SymbolKind.SECTION + assert _tree(found) == [ + ('title', SymbolKind.METADATA, 'T'), + ('schemas', section, '', [('A', SymbolKind.TYPE, 'string')]), + ('traits', section, ''), + ('/r', SymbolKind.RESOURCE, '', [ + ('uriParameters', section, ''), + ('get', SymbolKind.METHOD, '', [ + ('headers', section, ''), + ('200', SymbolKind.RESPONSE, '', [('body', section, '')]), + ]), + ]), + ] # fmt: skip + assert found[2].selection.line == _where(document, 'traits')[0] + + def test_a_section_a_template_supplied_is_not_the_methods(self, memory_workspace): + # The trait wrote `queryParameters:`; the method wrote `headers:` only. + document = ( + '#%RAML 1.0\ntitle: T\ntraits:\n paged:\n queryParameters:\n page: integer\n' + '/r:\n get:\n is: [paged]\n headers:\n X-Id: string\n' + ) + workspace, folder = _buffered(memory_workspace, {'api.raml': document}) + snapshot = workspace.snapshot(f'{folder}/api.raml') + resource = outline.document_symbols(snapshot, f'{folder}/api.raml')[-1] + (method,) = resource.children + assert [(child.name, child.selection.line) for child in method.children] == [ + ('is', _where(document, 'paged]')[0]), + ('headers', _where(document, 'headers:')[0]), + ] + + def test_the_parser_records_no_section_of_what_a_template_wrote(self, memory_workspace): + # The resource type's method and response are written in the template; + # recording their sections at each application would grow the model. + document = ( + '#%RAML 1.0\ntitle: T\nresourceTypes:\n rt:\n get:\n queryParameters:\n q: string\n' + ' responses:\n 200:\n body:\n application/json:\n' + '/a:\n type: rt\n/b:\n type: rt\n uriParameters: {}\n' + ) + workspace, folder = _buffered(memory_workspace, {'api.raml': document}) + raml = workspace.snapshot(f'{folder}/api.raml').raml + assert raml is not None + recorded = [(each.name, each.key.line) for each in raml.written_sections[f'{folder}/api.raml']] + assert recorded == [('resourceTypes', 3), ('uriParameters', _where(document, 'uriParameters')[0])] + + def test_a_types_facets_and_a_schemes_described_by_are_placed_at_their_keys(self, memory_workspace): + document = ( + '#%RAML 1.0\ntitle: T\ntypes:\n A:\n facets:\n unit: string\n' + 'securitySchemes:\n s:\n type: Basic Authentication\n describedBy:\n' + ' headers:\n Authorization: string\n' + ) + workspace, folder = _buffered(memory_workspace, {'api.raml': document}) + snapshot = workspace.snapshot(f'{folder}/api.raml') + _title, types, schemes = outline.document_symbols(snapshot, f'{folder}/api.raml') + (facets,) = types.children[0].children + (described,) = schemes.children[0].children + (headers,) = described.children + assert [(each.name, each.selection.line) for each in (facets, described, headers)] == [ + ('facets', _where(document, 'facets:')[0]), + ('describedBy', _where(document, 'describedBy:')[0]), + ('headers', _where(document, 'headers:')[0]), + ] + + def test_an_extension_places_what_it_restated_at_its_own_keys(self, memory_workspace): + # The merge keeps the master's `types:` and `/a:` keys; each document's + # own are recorded before it (docs/21 § 4). + api = '#%RAML 1.0\ntitle: T\ntypes:\n A: string\n/a:\n get:\n' + extension = '#%RAML 1.0 Extension\nextends: api.raml\ntypes:\n B: string\n/a:\n post:\n' + workspace, folder = _buffered(memory_workspace, {'api.raml': api, 'ext.raml': extension}) + found = outline.document_symbols(workspace.snapshot(f'{folder}/ext.raml'), f'{folder}/ext.raml') + assert [(each.name, each.selection.line, each.span.end_line) for each in found] == [ + ('types', _where(extension, 'types:')[0], _where(extension, 'B:')[0]), + ('/a', _where(extension, '/a:')[0], _where(extension, 'post:')[0]), + ] + own = outline.document_symbols(workspace.snapshot(f'{folder}/api.raml'), f'{folder}/api.raml') + assert [(each.name, each.selection.line) for each in own[1:]] == [ + ('types', _where(api, 'types:')[0]), + ('/a', _where(api, '/a:')[0]), + ] def test_a_symbol_spans_its_value_and_selects_its_name(self, parsed): snapshot, folder = parsed diff --git a/tests/unit/test_service_workspace.py b/tests/unit/test_service_workspace.py index 3fdcef6..8ef1ca9 100644 --- a/tests/unit/test_service_workspace.py +++ b/tests/unit/test_service_workspace.py @@ -233,6 +233,19 @@ def test_a_file_no_root_reads_is_parsed_alone(self, tmp_path): assert snapshot.root == f'{folder}/lib.raml' assert snapshot.error is None + def test_a_snapshot_keeps_an_open_buffers_text_not_a_second_copy(self, memory_workspace): + folder = path_to_file_uri(memory_workspace.root) + workspace = Workspace([folder]) + texts = {'api.raml': self.FILES['api.raml'], 'lib.raml': BOM + LIBRARY} + for name, text in texts.items(): + workspace.open(f'{folder}/{name}', text, 1) + snapshot = workspace.snapshot(f'{folder}/api.raml') + assert snapshot.raml is not None + kept = snapshot.raml.source_texts + assert kept[f'{folder}/api.raml'] is workspace.buffers[f'{folder}/api.raml'].text + # The parser drops the leading BOM, so that text is the parse's own. + assert kept[f'{folder}/lib.raml'] == LIBRARY + def test_an_os_error_is_keyed_by_its_errno(self, tmp_path, monkeypatch): """docs/11 § 6: an `OSError` past the parse reports a key, not the OS text.""" workspace, folder = _workspace(tmp_path, {'api.raml': API})