Skip to content

Commit e1809dc

Browse files
committed
add a new YAMLDocumentAggregator
Signed-off-by: Jiri Jaburek <comps@nomail.dom>
1 parent 2480746 commit e1809dc

5 files changed

Lines changed: 278 additions & 11 deletions

File tree

‎atex/aggregator/jsonl/README.md‎

Lines changed: 4 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -22,6 +22,7 @@ For example
2222

2323
["11.0@x86_64", "pass", "/ltp", "syscalls/alarm01", ["test.out"], null]
2424
["11.0@x86_64", "pass", "/ltp", "syscalls/socketpair02", ["server/test.out", "client/test.out"], null]
25+
["11.0@x86_64", "pass", "/ltp", null, [], "Suite version: 20260130"]
2526
```
2627

2728
- `aggregated/uploaded_files/`
@@ -52,10 +53,9 @@ The results use a top-level array (on each line) with a fixed item order:
5253
All these are strings except `files`, which is another (nested) array
5354
of strings.
5455

55-
Note that test name is explicitly given to `ingest()`, and subtest name comes
56-
from test artifacts (the `name` result key, which may be non-existent,
57-
indicating the result is relevant to the test itself, not a subtest).\
58-
See also [RESULTS.md](../../executor/fmf/RESULTS.md).
56+
- `platform` and `test name` are the strings given to `.ingest()`,
57+
- `status`, `subtest name`, `files` and `note` come from the ingested
58+
[Test Artifacts](../../executor)
5959

6060
Further:
6161
- If subtest name or note missing in test artifacts, a `null` item is used.

‎atex/aggregator/jsonl/jsonl.py‎

Lines changed: 9 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -12,10 +12,12 @@
1212
get_logger = util.get_loggers("atex.aggregator.jsonl")
1313

1414

15-
def _verbatim_move(src, dst):
16-
def copy_without_symlinks(src, dst):
17-
return shutil.copy2(src, dst, follow_symlinks=False)
18-
shutil.move(src, dst, copy_function=copy_without_symlinks)
15+
def verbatim_move(src, dst):
16+
return shutil.move(
17+
src,
18+
dst,
19+
copy_function=lambda src, dst: shutil.copy2(src, dst, follow_symlinks=False),
20+
)
1921

2022

2123
class JSONLinesAggregator(Aggregator):
@@ -73,7 +75,7 @@ def _move_test_files(test_files, target_dir):
7375
by the test, into the pre-computed `target_dir` location (inside
7476
a hierarchy of all files from all tests).
7577
"""
76-
_verbatim_move(test_files, target_dir)
78+
verbatim_move(test_files, target_dir)
7779

7880
def _gen_test_results(self, input_fobj, platform, test_name):
7981
"""
@@ -146,7 +148,7 @@ def ingest(self, platform, test_name, artifacts):
146148
# clean up the source test_results (Aggregator should 'mv', not 'cp')
147149
Path(artifacts_results).unlink()
148150

149-
# if the test_files dir is not empty
151+
# if the artifacts files directory is not empty
150152
if any(artifacts_files.iterdir()):
151153
platform_files.mkdir(exist_ok=True)
152154
# TODO: why does this work without .mkdir(target_test_files.parent) ?
@@ -192,7 +194,7 @@ def _move_test_files(self, test_files, target_dir):
192194

193195
# skip dirs, symlinks, device files, etc.
194196
if not src_path.is_file(follow_symlinks=False) or file_name in self.exclude:
195-
_verbatim_move(src_path, dst_path)
197+
verbatim_move(src_path, dst_path)
196198
continue
197199

198200
if self.suffix:

‎atex/aggregator/yamld/README.md‎

Lines changed: 102 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,102 @@
1+
> [!NOTE]
2+
> This describes a specific implementation of the abstract Aggregator API.
3+
> See also the [documentation of the generic API](..).
4+
5+
# YAML Document Aggregator
6+
7+
This Aggregator collects reported results into a single YAML file, each test
8+
result into a new `---` separated **YAML document**.
9+
10+
```yaml
11+
---
12+
first test result here
13+
---
14+
second result here
15+
---
16+
third here
17+
```
18+
19+
This allows efficient streamable reading as the reader has to only store one
20+
result in memory while parsing it.
21+
22+
Note that any subtest results the test may have reported are stored inside the
23+
one YAML document too, so this format is not suited for one test reporting
24+
millions of subtest results. Use [JSONLinesAggregator](../jsonl) for that.
25+
26+
## Format
27+
28+
- `platform` and `name` are the strings given to `.ingest()`,
29+
- `status`, `files`, `note` and `subtests` come from the ingested
30+
[Test Artifacts](../../executor)
31+
32+
For example,
33+
34+
- `aggregated/results.yaml`
35+
36+
```yaml
37+
---
38+
platform: 9.8@x86_64
39+
name: /some/test
40+
status: pass
41+
---
42+
platform: 10.2@s390x
43+
name: /unit/syscalls
44+
status: pass
45+
files:
46+
- full_output.txt
47+
subtests:
48+
- name: accept
49+
status: pass
50+
files:
51+
- test.txt
52+
- name: connect
53+
status: fail
54+
files:
55+
- test.txt
56+
note: Got errno: ECONNABORTED
57+
- name: open
58+
status: warn
59+
files:
60+
- test.txt
61+
---
62+
platform: 11.0@x86_64
63+
name: /ltp
64+
status: pass
65+
note: 'Suite version: 20260130'
66+
subtests:
67+
- name: syscalls/alarm01
68+
status: pass
69+
files:
70+
- test.out
71+
- name: syscalls/socketpair02
72+
status: pass
73+
files:
74+
- server/test.out
75+
- client/test.out
76+
```
77+
78+
- `aggregated/uploaded_files/`
79+
80+
```
81+
/10.2@s390x/unit/syscalls/accept/test.txt
82+
/10.2@s390x/unit/syscalls/connect/test.txt
83+
/10.2@s390x/unit/syscalls/open/test.txt
84+
/10.2@s390x/unit/syscalls/full_output.txt
85+
86+
/11.0@x86_64/ltp/syscalls/alarm01/test.out
87+
/11.0@x86_64/ltp/syscalls/socketpair02/server/test.out
88+
/11.0@x86_64/ltp/syscalls/socketpair02/client/test.out
89+
```
90+
91+
Further:
92+
- If subtest name or note missing in test artifacts, it is omitted in the YAML
93+
(eg. `note: null` never appears).
94+
- If `testout` is present inside test artifacts (in the result for the test
95+
itself), it is prepended to the list of `files`.
96+
97+
## Examples
98+
99+
```python
100+
with YAMLDocumentAggregator("results.yaml", "uploaded_files") as aggr:
101+
aggr.ingest("9.8@x86_64", "/some/test", test_artifacts_dir)
102+
```

‎atex/aggregator/yamld/__init__.py‎

Lines changed: 7 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,7 @@
1+
from .yamld import ( # noqa: F401, I001
2+
YAMLDocumentAggregator,
3+
)
4+
5+
__all__ = (
6+
"YAMLDocumentAggregator",
7+
)

‎atex/aggregator/yamld/yamld.py‎

Lines changed: 156 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,156 @@
1+
import json
2+
import threading
3+
from pathlib import Path
4+
5+
import yaml
6+
7+
from ... import util
8+
from .. import Aggregator, AggregatorError
9+
from ..jsonl.jsonl import verbatim_move
10+
11+
get_logger = util.get_loggers("atex.aggregator.yamld")
12+
13+
14+
class YamlDocumentAggregator(Aggregator):
15+
"""
16+
- `target` is a string/Path to a `.yaml` file for all ingested results
17+
to be aggregated (written) to.
18+
19+
- `files` is a string/Path of the top-level parent for all per-platform
20+
/ per-test files uploaded by tests.
21+
22+
- `allow_duplicate` permits any one test name to be ingested more than
23+
once, appending ` (1)` to the second test name entry, ` (2)` to the
24+
third, etc.
25+
"""
26+
27+
def __init__(self, target, files, *, allow_duplicate=False):
28+
self.lock = threading.RLock()
29+
self.logger = get_logger()
30+
31+
self.target = Path(target)
32+
self.files = Path(files)
33+
self.allow_duplicate = allow_duplicate
34+
self.seen_tests = {}
35+
self.target_fobj = None
36+
37+
def start(self):
38+
self.logger.debug(f"starting: {self}")
39+
40+
if self.target.exists(follow_symlinks=False):
41+
raise FileExistsError(f"{self.target} already exists")
42+
self.target_fobj = open(self.target, "w")
43+
44+
if self.files.exists(follow_symlinks=False):
45+
raise FileExistsError(f"{self.files} already exists")
46+
self.files.mkdir()
47+
48+
def stop(self):
49+
self.logger.debug(f"stopping: {self}")
50+
51+
if self.target_fobj:
52+
self.target_fobj.close()
53+
self.target_fobj = None
54+
55+
def ingest(self, platform, test_name, artifacts):
56+
unique_id = (platform, test_name)
57+
with self.lock:
58+
if unique_id in self.seen_tests:
59+
if not self.allow_duplicate:
60+
raise AggregatorError(
61+
f"'{test_name}' was already ingested once for '{platform}'",
62+
)
63+
else:
64+
test_name = f"{test_name} ({self.seen_tests[unique_id]})"
65+
self.seen_tests[unique_id] += 1
66+
else:
67+
self.seen_tests[unique_id] = 1
68+
69+
self.logger.info(f"ingesting '{platform}' / '{test_name}' from '{artifacts}'")
70+
71+
artifacts = Path(artifacts)
72+
artifacts_results = artifacts / "results"
73+
artifacts_files = artifacts / "files"
74+
75+
if not artifacts_results.exists(follow_symlinks=False):
76+
raise FileNotFoundError(f"{artifacts_results} does not exist")
77+
78+
platform_files = self.files / util.normalize_path(platform)
79+
target_test_files = platform_files / util.normalize_path(test_name)
80+
if target_test_files.exists(follow_symlinks=False):
81+
raise FileExistsError(f"{target_test_files} already exists for {test_name}")
82+
83+
# any None or empty values are deleted later,
84+
# to preserve dict insertion order with these on top
85+
document = {
86+
"platform": platform,
87+
"name": test_name,
88+
"status": None,
89+
"note": None,
90+
"files": [],
91+
"subtests": [],
92+
}
93+
94+
with open(artifacts_results) as f:
95+
for raw_line in f:
96+
result_line = json.loads(raw_line)
97+
98+
# if it is a subtest, add it to subtests
99+
if name := result_line.get("name"):
100+
subtest = {"name": name}
101+
if status := result_line.get("status"):
102+
subtest["status"] = status
103+
if files := result_line.get("files"):
104+
subtest["files"] = files
105+
if note := result_line.get("note"):
106+
subtest["note"] = note
107+
document["subtests"].append(subtest)
108+
109+
# update document for the test itself
110+
else:
111+
if status := result_line.get("status"):
112+
document["status"] = status
113+
if note := result_line.get("note"):
114+
document["note"] = note
115+
116+
file_names = []
117+
# process the file specified by the 'testout' key
118+
if "testout" in result_line:
119+
file_names.append(result_line["testout"])
120+
# process any additional files in the 'files' key
121+
if files := result_line.get("files"):
122+
file_names += files
123+
if file_names:
124+
document["files"] += file_names
125+
126+
if document["status"] is None:
127+
del document["status"]
128+
if document["note"] is None:
129+
del document["note"]
130+
if not document["files"]:
131+
del document["files"]
132+
if not document["subtests"]:
133+
del document["subtests"]
134+
135+
with self.lock:
136+
yaml.dump(
137+
document,
138+
self.target_fobj,
139+
explicit_start=True,
140+
default_flow_style=False,
141+
sort_keys=False,
142+
)
143+
self.target_fobj.flush()
144+
145+
# clean up the source test_results (Aggregator should 'mv', not 'cp')
146+
Path(artifacts_results).unlink()
147+
148+
# if the artifacts files directory is not empty
149+
if any(artifacts_files.iterdir()):
150+
platform_files.mkdir(exist_ok=True)
151+
# TODO: why does this work without .mkdir(target_test_files.parent) ?
152+
verbatim_move(artifacts_files, target_test_files)
153+
154+
def __str__(self):
155+
class_name = self.__class__.__name__
156+
return f"{class_name}({str(self.target)}, {str(self.files)})"

0 commit comments

Comments
 (0)