Repository navigation
Expand file tree
/
Copy pathapi.py
More file actions
2237 lines (2020 loc) · 85.5 KB
/
Copy pathapi.py
File metadata and controls
2237 lines (2020 loc) · 85.5 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
# DO NOT EDIT — generated by scripts/regenerate_sync.py from
# the canonical async source in polyswarm_api/aio/.
# Edit aio/<file>.py then rerun the script.
"""PolySwarm API client.
The canonical version of this file lives at
``polyswarm_api/aio/api.py``; the sync mirror at
``polyswarm_api/api.py`` is generated from it by
``scripts/regenerate_sync.py``. Edit only the canonical (async) file —
the sync side is mechanically derived.
Public surface::
from polyswarm_api import PolyswarmAPI # sync
from polyswarm_api.aio import PolySwarmAsyncAPI # async
"""
import time
import io
import logging
from polyswarm_api import exceptions, refang, resources, settings
from polyswarm_api.core import PolyswarmRequest, _as_result_bound
from .session import PolyswarmSession
logger = logging.getLogger(__name__)
__all__ = ["PolyswarmAPI"]
class PolyswarmAPI:
"""Interface to the PolySwarm API.
The canonical client lives at ``polyswarm_api.aio`` (async, uses
``httpx.AsyncClient``); the sync mirror at ``polyswarm_api`` is
generated from it and uses ``httpx.Client``. Both expose the same
method names and return shapes — sync callers receive values /
iterators directly; async callers ``await`` / ``async for``.
"""
# Pagination safety bound: ``_consume_results`` never walks more than this
# many pages, even if the server keeps reporting ``has_more`` (or returns a
# cursor that doesn't advance). A misbehaving endpoint must not be able to
# hang a client in an unbounded loop; the cap is far above any legitimate
# result set (50/page => 500k results).
_MAX_PAGES = 10_000
def __init__(
self,
key: str | None = None,
uri: str | None = None,
community: str | None = None,
timeout: float | None = None,
verify: bool = True,
*,
session: PolyswarmSession | None = None,
refang_iocs: bool = False,
**httpx_kwargs,
):
key_masked = "******" + (key[-4:] if key and len(key) > 16 else "")
logger.info(
"Creating %s instance | api-key: %s, api-uri: %s, community: %s",
type(self).__name__,
key_masked,
uri,
community,
)
self.uri = uri or settings.DEFAULT_GLOBAL_API
self.community = community or settings.DEFAULT_COMMUNITY
self.timeout = timeout or settings.DEFAULT_HTTP_TIMEOUT
self.verify = verify
# Refang defanged URL / domain / IP inputs (``hxxps[:]//evil[.]com``)
# before building a request. Opt-in: the default preserves the 4.5.0
# behaviour of sending every input verbatim. See ``polyswarm_api.refang``
# and ``_refang`` below for exactly which inputs are touched.
self.refang_iocs = refang_iocs
self._engines = None
# Either accept a pre-built session (customization point) or
# build the default from ``key``. Passing both is ambiguous.
if session is not None:
if httpx_kwargs:
raise exceptions.InvalidValueException(
"session= and httpx_kwargs are mutually exclusive — "
"configure the session directly.",
)
self.session = session
else:
self.session = PolyswarmSession(
key,
retries=settings.DEFAULT_RETRIES,
verify=verify,
timeout=self.timeout,
**httpx_kwargs,
)
def __repr__(self):
clsname = f"{type(self).__module__}.{type(self).__name__}"
attrs = (
f"uri={self.uri!r}, community={self.community!r}, timeout={self.timeout!r}"
)
return f"<{clsname}({attrs}) at 0x{id(self):x}>"
def _refang(self, value):
"""Refang one URL / domain / IP input when ``refang_iocs`` is on.
Applied only to arguments that name a network indicator (a URL to
search or submit, an ``ips=`` / ``urls=`` / ``domains=`` entry, an IoC
``ip`` / ``domain``, a known-host ``host``) — never to a free-form
metadata query or a hash.
Anything that is not a defanged network IoC is returned unchanged.
"""
if not self.refang_iocs:
return value
return refang.refang_ioc(value)
def _refang_all(self, values):
"""``_refang`` over a list argument; ``None`` / empty pass through."""
if not values or not self.refang_iocs:
return values
if isinstance(values, str):
return self._refang(values)
return [self._refang(v) for v in values]
def close(self):
"""Close the underlying HTTP client."""
self.session.close()
def __enter__(self):
return self
def __exit__(self, *exc):
self.close()
# ── Transport hooks ─────────────────────────────────────────────
#
# ``_single`` / ``_paginate`` accept either a ``PolyswarmRequest``
# descriptor returned by a resource classmethod or a
# request-parameters dict (inline endpoint bodies).
def _to_request(self, request, result_parser=None, parser_kwargs=None):
"""Coerce ``request`` into a ``PolyswarmRequest`` descriptor.
A ``dict`` is treated as request-parameters ({method, url, params,
json, headers, data, files, ...}); anything else is assumed to
already be a descriptor (and is returned untouched).
"""
if isinstance(request, PolyswarmRequest):
return request
if not isinstance(request, dict):
raise exceptions.InvalidValueException(
f"expected dict or PolyswarmRequest, got {type(request).__name__}",
)
rp = dict(request)
method = rp.pop("method")
url = rp.pop("url")
return PolyswarmRequest(
api=self,
method=method,
url=url,
params=rp.pop("params", None),
json=rp.pop("json", None),
headers=rp.pop("headers", None),
content=rp.pop("content", None),
data=rp.pop("data", None),
files=rp.pop("files", None),
timeout=rp.pop("timeout", None),
result_parser=result_parser,
parser_kwargs=parser_kwargs or {},
)
def _single(self, request, result_parser=None, **kwargs):
req = self.session.execute(self._to_request(request, result_parser, kwargs))
if req._paginated:
return self._consume_results(req)
return req._result
def _paginate(self, request, result_parser=None, **kwargs):
req = self.session.execute(self._to_request(request, result_parser, kwargs))
if req._paginated:
for item in self._consume_results(req):
yield item
else:
result = req._result
if isinstance(result, list):
for item in result:
yield item
elif result is not None:
yield result
def _consume_results(self, request):
# Bounded by _MAX_PAGES, and stops if the offset/cursor fails to advance,
# so a server that leaves has_more set can't loop the client forever.
seen_offsets = set()
for _page in range(self._MAX_PAGES):
try:
for item in request._result:
yield item
except TypeError:
yield request._result
return
if not request.has_more:
return
offset = request.offset
# A cursor that can't advance ends pagination: absent/None (re-sending
# offset=None is byte-identical to the page just fetched — e.g. a
# live-feed envelope with has_more but no offset) or a repeat of one
# already seen. Either way the next page would be identical.
if offset is None or offset in seen_offsets:
logger.warning(
"Stopping pagination: cursor %r did not advance while "
"has_more was still set.",
offset,
)
return
seen_offsets.add(offset)
request = self._next_page(request)
logger.warning(
"Stopping pagination at the %d-page safety cap.", self._MAX_PAGES
)
def _next_page(self, request):
"""Build the next-page descriptor by cloning ``request`` with
updated ``params['offset' | 'limit']`` and dispatch it through
the session.
Re-sends ``request.input_json`` — the descriptor's send body.
``request.json`` is the *response* body after execution, so it
must never be re-sent as the next request's body.
"""
params = request.params
if params is None:
params = {}
if isinstance(params, dict):
new_params = dict(params)
new_params["offset"] = request.offset
new_params["limit"] = request.limit
else:
new_params = [p for p in params if p[0] != "offset" and p[0] != "limit"]
new_params.extend([("offset", request.offset), ("limit", request.limit)])
next_req = PolyswarmRequest(
api=request.api,
method=request.method,
url=request.url,
params=new_params,
json=request.input_json,
headers=request.headers,
content=request.content,
data=request.data,
files=request.files,
timeout=request.timeout,
result_parser=request.result_parser,
parser_kwargs=request.parser_kwargs,
)
return self.session.execute(next_req)
# ── Engines ──────────────────────────────────────────────────
def engines(self):
"""Return the cached engine listing, refreshing it on first use.
``engines`` is a method, not the cached *property* it was in 3.x,
because the refresh performs I/O — on the async client it must be
awaited (the property->method change is a documented 4.0 break). Call
``refresh_engine_cache()`` to force a refresh.
"""
if not self._engines:
self.refresh_engine_cache()
return self._engines
def refresh_engine_cache(self):
"""Refresh the cached engine listing."""
engine_list = []
for engine in self._paginate(resources.Engine.list(self)):
engine_list.append(engine)
if not engine_list:
raise exceptions.InvalidValueException("Received empty engines listing")
self._engines = engine_list
def _parse_rule(self, rule):
"""Pure helper: normalise a rule argument into (ruleset, rule_id)."""
if isinstance(rule, str):
rule, rule_id = resources.YaraRuleset(dict(yara=rule), api=self), None
elif isinstance(rule, (resources.YaraRuleset, int)):
rule, rule_id = None, rule
else:
raise exceptions.InvalidValueException(
"Either yara or rule_id must be provided."
)
return rule, rule_id
# ── Endpoint methods (canonical) ────────────────────────────
def metadata_mapping(self):
"""Return the ``MetadataMapping`` describing available metadata
field names and types."""
logger.info("Retrieving the metadata mapping")
return self._single(
{
"method": "GET",
"url": f"{self.uri}{resources.MetadataMapping.RESOURCE_ENDPOINT}",
},
result_parser=resources.MetadataMapping,
)
def metadata_field_properties_write(
self, field_path, description, example=None, category=None, aliases=None
):
"""Upsert a metadata field properties entry (the single write path).
:param field_path: Dotted ES leaf path (e.g. 'polyunite.malware_family').
:param description: Human-readable description of the field.
:param example: Optional example search string.
:param category: Optional category grouping the field belongs to.
:param aliases: Optional list of friendly-name shortcuts.
:return: A ``MetadataFieldProperties`` resource.
"""
logger.info("Writing metadata field properties %s", field_path)
# Omit unset optionals from the body (3.x dropped None-valued fields in
# _params; the rest of this client follows the same convention). Sending
# explicit nulls is a request-shape regression.
body = {"field_path": field_path, "description": description}
if example is not None:
body["example"] = example
if category is not None:
body["category"] = category
if aliases is not None:
body["aliases"] = aliases
return self._single(
{
"method": "POST",
"url": f"{self.uri}{resources.MetadataFieldProperties.RESOURCE_ENDPOINT}",
"json": body,
},
result_parser=resources.MetadataFieldProperties,
)
def metadata_field_properties_get(self, field_path):
"""Get a metadata field properties entry by ``field_path``."""
logger.info("Getting metadata field properties %s", field_path)
return self._single(
{
"method": "GET",
"url": f"{self.uri}{resources.MetadataFieldProperties.RESOURCE_ENDPOINT}",
"params": {"field_path": field_path},
},
result_parser=resources.MetadataFieldProperties,
)
def metadata_field_properties_delete(self, field_path):
"""Delete a metadata field properties entry."""
logger.info("Deleting metadata field properties %s", field_path)
return self._single(
{
"method": "DELETE",
"url": f"{self.uri}{resources.MetadataFieldProperties.RESOURCE_ENDPOINT}",
"params": {"field_path": field_path},
},
result_parser=resources.MetadataFieldProperties,
)
def metadata_field_properties_list(self):
"""Iterate all metadata field properties entries."""
logger.info("Listing metadata field properties")
for item in self._paginate(
{
"method": "GET",
"url": f"{self.uri}{resources.MetadataFieldProperties.RESOURCE_ENDPOINT}/list",
},
result_parser=resources.MetadataFieldProperties,
):
yield item
def search(self, hash_, hash_type=None):
"""
Search for the latest scans matching the given hash and hash_type.
:param hash_: A Hashable object (Artifact, local.LocalArtifact, Hash) or hex-encoded SHA256/SHA1/MD5
:param hash_type: Hash type of the provided hash_. Will attempt to auto-detect if not explicitly provided.
:return: Generator of ArtifactInstance resources
"""
logger.info("Searching for hash %s", hash_)
hash_ = resources.Hash.from_hashable(hash_, hash_type=hash_type)
for item in self._paginate(
resources.ArtifactInstance.search_hash(self, hash_.hash, hash_.hash_type)
):
yield item
def search_url(self, url):
"""
Search for the latest scan matching the given url.
:param url: A url to be searched by exact match
:return: Generator of ArtifactInstance resources
"""
url = self._refang(url)
logger.info("Searching for url %s", url)
for item in self._paginate(resources.ArtifactInstance.search_url(self, url)):
yield item
def search_scans(self, hash_):
"""
Search for all scans ever made matching the given sha256.
:param hash_: A Hashable object (Artifact, local.LocalArtifact, Hash) or hex-encoded SHA256
:return: Generator of ArtifactInstance resources
"""
logger.info("Searching for scans %s", hash_)
hash_ = resources.Hash.from_hashable(hash_, hash_type="sha256")
for item in self._paginate(
resources.ArtifactInstance.list_scans(self, hash_.hash)
):
yield item
def search_by_metadata(
self, query, include=None, exclude=None, ips=None, urls=None, domains=None
):
"""
Search artifacts by metadata
:param query: A query string
:param include: A list of fields to be included in the result (.* wildcards are accepted)
:param exclude: A list of fields to be excluded from the result (.* wildcards are accepted)
:return: Generator of ArtifactInstance resources
"""
ips, urls, domains = (
self._refang_all(ips),
self._refang_all(urls),
self._refang_all(domains),
)
logger.info("Searching for metadata %s", query)
for item in self._paginate(
resources.Metadata.get(
self,
query=query,
community=self.community,
include=include,
exclude=exclude,
ips=ips,
urls=urls,
domains=domains,
)
):
yield item
def iocs_by_hash(self, hash_type, hash_value, hide_known_good=False, beta=False):
"""
Retrieve IOCs by artifact hash
:param hash_type: Hash type of the provided hash_
:param hash_value: A list of fields to be included in the result (.* wildcards are accepted)
:return: Generator of IOC resources
"""
logger.info("Getting IOCs by hash %s:%s", hash_type, hash_value)
for item in self._paginate(
resources.IOC.iocs_by_hash(
self, hash_value, hash_type, hide_known_good=hide_known_good, beta=beta
)
):
yield item
def search_by_ioc(
self, ip=None, domain=None, ttp=None, imphash=None, with_artifacts=False
):
"""
Search artifacts by IOC (ip, domain, ttp, or imphash)
:param ip: ip address to search by
:param domain: domain address to search by
:param ttp: ttp to search by
:param imphash: ImpHash to search by
:param with_artifacts: True yields a Metadata resource per matching artifact
(a metadata-search row trimmed to a field set the server owns: artifact.*,
the scan summary, ssdeep/tlsh and the malware family) instead of its bare
sha256. With no ip/domain/ttp/imphash the server refuses it with a 400
(a typed exception); without it a bare call behaves as it always has.
Needs a server that supports the parameter:
an older one ignores it and answers bare sha256 strings, which fail to
parse as Metadata (TypeError).
:return: Generator of IOC resources whose ``json`` is a sha256 string, or of
Metadata resources when ``with_artifacts`` is True
"""
ip, domain = self._refang(ip), self._refang(domain)
logger.info(
"Searching by ioc %s",
dict(
ip=ip,
domain=domain,
ttp=ttp,
imphash=imphash,
with_artifacts=with_artifacts,
),
)
for item in self._paginate(
resources.IOC.ioc_search(
self,
ip=ip,
domain=domain,
ttp=ttp,
imphash=imphash,
with_artifacts=with_artifacts,
)
):
yield item
def check_known_hosts(self, ips=[], domains=[]):
"""
Check if ip addresses or domains are known.
:param ips
:param domains
:return: Generator of IOC resources
"""
ips, domains = self._refang_all(ips), self._refang_all(domains)
logger.info("Checking known hosts ips: %s, domains: %s", ips, domains)
for item in self._paginate(resources.IOC.check_known_hosts(self, ips, domains)):
yield item
def add_known_good_host(self, type, source, host):
"""
Add a known good ip or domain.
:param type
:param source
:param host
:return: IOC resource
"""
host = self._refang(host)
logger.info("Creating known good ioc %s %s %s", type, host, source)
return self._single(resources.IOC.create_known_good(self, type, host, source))
def add_known_bad_host(self, type, source, host):
"""
Add a known bad ip or domain.
:param type
:param source
:param host
:return: IOC resource
"""
host = self._refang(host)
logger.info("Creating known bad ioc %s %s %s", type, host, source)
return self._single(resources.IOC.create_known_bad(self, type, host, source))
def update_known_good_host(self, id, type, source, host, good):
"""
Update a known ip or domain.
:param type
:param source
:param host
:return: IOC resource
"""
host = self._refang(host)
logger.info("Updating known good ioc %s %s %s %s", id, type, host, source)
return self._single(
resources.IOC.update_known_good(self, id, type, host, source, good)
)
def delete_known_good_host(self, id):
logger.info("Deleting known good ioc %s", id)
return self._single(resources.IOC.delete_known_good(self, id))
def lookup(self, scan):
"""
Lookup a scan by Scan id.
:param scan: The Scan UUID to lookup
:return: An ArtifactInstance resource
"""
logger.info("Lookup scan %s", int(scan))
return self._single(resources.ArtifactInstance.lookup_uuid(self, scan))
def rescan(self, hash_, hash_type=None, scan_config=None):
"""
Rescan a file based on and existing hash in the Polyswarm platform
:param hash_: Hashable object (Artifact, local.LocalArtifact, or Hash) or hex-encoded SHA256/SHA1/MD5
:param hash_type: Hash type of the provided hash_. Will attempt to auto-detect if not explicitly provided.
:param scan_config: The scan configuration to be used, e.g.: "default", "more-time", "most-time"
:return: A ArtifactInstance resources
"""
logger.info("Rescan hash %s", hash_)
hash_ = resources.Hash.from_hashable(hash_, hash_type=hash_type)
return self._single(
resources.ArtifactInstance.rescan(
self, hash_.hash, hash_.hash_type, scan_config=scan_config
)
)
def rescan_id(self, scan, scan_config=None):
"""
Re-execute a new scan based on an existing scan.
:param scan: Id of the existing scan
:param scan_config: The scan configuration to be used, e.g.: "default", "more-time", "most-time"
:return: A ArtifactInstance resource
"""
logger.info("Rescan id %s", int(scan))
return self._single(
resources.ArtifactInstance.rescan_id(self, scan, scan_config=scan_config)
)
def live_start(self, rule_id):
"""
Create a new live hunt_id, and replace the currently running YARA rules.
:param rule_id: Yara ruleset id
:return: The ruleset with the associated live hunt
"""
logger.info("Create live hunt for rule id %s", rule_id)
return self._single(resources.LiveYaraRuleset.create(self, rule_id=rule_id))
def live_stop(self, rule_id):
"""
Stop a live hunt.
:param rule_id: Yara ruleset id
:return: The ruleset without an associate live hunt
"""
logger.info("Delete live hunt for rule id %s", rule_id)
return self._single(resources.LiveYaraRuleset.delete(self, rule_id=rule_id))
def live_feed(
self,
since=None,
rule_name=None,
family=None,
polyscore_lower=None,
polyscore_upper=None,
community=None,
livescan_id=None,
max_results=None,
):
"""
Get live hunts feed
:param since: Window in SECONDS (this said "minutes" in earlier releases and was
wrong). Absent or 0 means no time filter at all.
:param rule_name: Filter hunt results on the provided rule name (exact match).
:param family: Filter hunt results based on the family name (exact match).
:param polyscore_lower: Polyscore lower bound for the hunt results.
:param polyscore_upper: Polyscore upper bound for the hunt results.
:param community: Community to retrieve live results from, or public/private.
:param livescan_id: Scope the feed to one live hunt's results.
:param max_results: Total results to yield, not a page size — paging
continues in the server's own chunks until the total is reached.
None, 0 or a negative means no bound: every page, as before.
:return: Generator of HuntResult resources
"""
bound = _as_result_bound(max_results)
yielded = 0
for item in self._paginate(
resources.LiveHuntResult.list(
self,
since=since,
rule_name=rule_name,
family=family,
polyscore_lower=polyscore_lower,
polyscore_upper=polyscore_upper,
livescan_id=livescan_id,
community=community or self.community,
)
):
yield item
yielded += 1
if bound is not None and yielded >= bound:
return
def live_feed_delete(self, result_ids):
"""
Delete live feed results
:param result_ids: Live Feed Result IDs
:return: The deleted LiveHuntResult resources
"""
logger.info("Delete live results: %s", result_ids)
try:
return self._single(
resources.LiveHuntResultList.delete(self, result_ids=result_ids)
)
except exceptions.NoResultsException:
return None
def live_result(self, result_id):
"""
Get yara ruleset for the live hunt result
:param result_id: Live result id
:return: A LiveHuntResult resource
"""
return self._single(resources.LiveHuntResult.get(self, id=result_id))
def historical_create(self, rule=None, ruleset_name=None):
"""
Run a new historical hunt.
:param rule: YaraRuleset object or string containing YARA rules to install
:param ruleset_name: Name of the ruleset.
:return: The created Hunt resource
"""
logger.info("Create historical hunt %s", rule)
rule, rule_id = self._parse_rule(rule)
return self._single(
resources.HistoricalHunt.create(
self,
yara=rule.yara if rule else None,
rule_id=rule_id,
ruleset_name=ruleset_name,
community=self.community,
)
)
def historical_get(self, hunt=None):
"""
Get a historical hunt.
:param hunt: Hunt ID
:return: The Hunt resource
"""
logger.info("Get historical hunt %s", hunt)
return self._single(
resources.HistoricalHunt.get(self, id=hunt, community=self.community)
)
def historical_update(self, hunt):
"""
Cancel a historical hunt
:param hunt: The historical hunt id
:return: The deleted HistoricalHunt resource
"""
logger.info("Deleting historical hunt %s", hunt)
return self._single(
resources.HistoricalHunt.update(self, id=hunt, community=self.community)
)
def historical_delete(self, hunt):
"""
Delete a historical hunts.
:param hunt: Hunt ID
:return: The deleted Hunt resource
"""
logger.info("Delete historical hunt %s", hunt)
return self._single(
resources.HistoricalHunt.delete(self, id=hunt, community=self.community)
)
def historical_list(self, since=None):
"""
List all historical hunts
:return: Generator of Hunt resources
"""
logger.info("List historical hunts since: %s", since)
for item in self._paginate(
resources.HistoricalHunt.list(self, since=since, community=self.community)
):
yield item
def historical_result(self, result_id):
"""
Get historical hunt result
:param result_id: Historical result id
:return: HistoricalHuntResult resource
"""
return self._single(
resources.HistoricalHuntResult.get(
self, id=result_id, community=self.community
)
)
def historical_results(
self,
hunt=None,
rule_name=None,
family=None,
polyscore_lower=None,
polyscore_upper=None,
community=None,
):
"""
Get results from a historical hunt
:param hunt: ID of the hunt (None if latest hunt results are desired)
:param rule_name: Filter hunt results on the provided rule name (exact match).
:param family: Filter hunt results based on the family name (exact match).
:param polyscore_lower: Polyscore lower bound for the hunt results.
:param polyscore_upper: Polyscore upper bound for the hunt results.
:param community: Community to retrieve live results from, or public/private.
:return: Generator of HuntResult resources
"""
logger.info("List historical results for hunt: %s", hunt)
for item in self._paginate(
resources.HistoricalHuntResultList.get(
self,
id=hunt,
rule_name=rule_name,
family=family,
community=community or self.community,
polyscore_lower=polyscore_lower,
polyscore_upper=polyscore_upper,
)
):
yield item
def historical_results_delete(self, result_ids):
"""
Delete historical scan results
:param result_ids: Historical Hunt Result IDs
:return: The deleted HuntResult resources
"""
logger.info("Delete historical results: %s", result_ids)
return self._single(
resources.HistoricalHuntResultList.delete(
self, result_ids=result_ids, community=self.community
)
)
def historical_delete_list(self, historical_ids):
"""
Delete historical hunts.
:param historical_ids: Historical Hunt IDs
:return: The deleted Hunt resource
"""
logger.info("Delete historical hunts %s", historical_ids)
return self._single(
resources.HistoricalHuntList.delete(
self, historical_ids=historical_ids, community=self.community
)
)
def ruleset_create(self, name, rules, description=None):
"""
Create a Yara Ruleset from the provided rules with the given name in the polyswarm platform.
:param name: Name of the ruleset
:param rules: Yara rules as a string
:param description: Description of the ruleset
:return: A YaraRuleset resource
"""
logger.info("Create ruleset %s: %s", name, rules)
rules = resources.YaraRuleset(
dict(
name=name, description=description, yara=rules, community=self.community
),
api=self,
)
return self._single(
resources.YaraRuleset.create(
self, yara=rules.yara, name=rules.name, description=rules.description
)
)
def ruleset_get(self, ruleset_id=None):
"""
Retrieve a YaraRuleset from the polyswarm platform by its Id.
:param ruleset_id: Id of the ruleset
:return: A YaraRuleset resource
"""
logger.info("Get ruleset %s", ruleset_id)
return self._single(
resources.YaraRuleset.get(self, id=ruleset_id, community=self.community)
)
def ruleset_update(self, ruleset_id, name=None, rules=None, description=None):
"""
Update an existing YaraRuleset in the polyswarm platform by its Id.
:param ruleset_id: Id of the ruleset
:param name: New name of the ruleset
:param rules: New yara rules as a string
:param description: New description of the ruleset
:return: The updated YaraRuleset resource
"""
logger.info("Update ruleset %s", ruleset_id)
return self._single(
resources.YaraRuleset.update(
self,
id=ruleset_id,
name=name,
yara=rules,
description=description,
community=self.community,
)
)
def ruleset_delete(self, ruleset_id):
"""
Delete a YaraRuleset from the polyswarm platform by its Id.
:param ruleset_id: Id of the ruleset
:return: A YaraRuleset resource
"""
logger.info("Delete ruleset %s", ruleset_id)
return self._single(
resources.YaraRuleset.delete(self, id=ruleset_id, community=self.community)
)
def ruleset_list(
self,
name=None,
status=None,
favorites_only=None,
has_new_results=None,
sort=None,
exclude_favorites=None,
):
"""
List all YaraRulesets for the current account.
All filters are optional and conjunctive:
:param name: Case-insensitive substring match on the ruleset name.
:param status: 'active' returns only rulesets whose live hunt is
currently running.
:param favorites_only: True returns only favorited rulesets.
:param exclude_favorites: True returns only the rulesets that are NOT
favorited — the inverse of ``favorites_only``, and refused together
with it (a contradiction, answered with an error rather than an
empty list). It exists for clients that render the favorites as
their own list: the favorites are a separate, unpaginated fetch
bounded by the account's budget, so leaving them in the paginated
list too makes a page either repeat a row or come back short.
Appended to the signature rather than placed beside
``favorites_only`` so a positional caller keeps working.
:param has_new_results: True returns only rulesets whose stored
new-results counter is positive. The counter (and its window) is
maintained server-side by a scheduled refresh; rows carry it as
``new_results_count`` with ``new_results_counted_at`` marking when
it was last refreshed. There is no per-request window parameter.
:param sort: ``'active_first'`` returns the rulesets that carry a live
hunt link first, newest first within each block. Default (None) is
newest first. "Newest first" is the server's own insertion key, NOT
the ``id`` on the rows you get back — that one is unique but
unordered, so dedupe with it and never resume or bound a walk with
it. Applied SERVER-side, across pages — the list is
keyset-paginated, so a client-side sort would only ever reorder one
page; the SDK never re-orders rows. Reuse a page's ``offset`` only
with the same ``sort``: the server refuses a cursor minted under
the other order.
Two server-side properties of that key, neither of them SDK
behaviour. It ranks on the stored link, which is a WIDER predicate
than the one ``livescan_id`` is rendered under: a legacy row whose
hunt was stopped without clearing the link ranks in the leading
block while still serializing ``livescan_id`` as ``None``. Read the
field to decide whether a ruleset is running; never the position.
And the key is MUTABLE, unlike that default: a ruleset whose
live hunt stops mid-walk falls back into the idle block below the
cursor and is yielded twice, and one started mid-walk moves above
the cursor and is skipped for the rest of that walk. That is a
property of the walk, so starting fresh from the first page does
not avoid it. This generator streams pages and does not dedupe —
dedupe by ``id`` if you consume more than one page.
:return: A generator of YaraRuleset resources
"""
logger.info("List rulesets")
for item in self._paginate(
resources.YaraRuleset.list(
self,
name=name,
status=status,
favorites_only=favorites_only,
has_new_results=has_new_results,
sort=sort,
exclude_favorites=exclude_favorites,
community=self.community,
)
):
yield item
def ruleset_favorite(self, ruleset_id, favorite=True):
"""
Favorite or unfavorite a YaraRuleset. Idempotent; works while a live
hunt is running. Favorites are shared by the whole team and capped
(the response carries favorites_used / favorites_limit); when the
budget is exhausted the server refuses with a machine-readable
FAVORITE_LIMIT error.
:param ruleset_id: Id of the ruleset
:param favorite: True to star, False to unstar
:return: A YaraRulesetFavorite resource
"""
logger.info(
"%s ruleset %s", "Favorite" if favorite else "Unfavorite", ruleset_id
)
return self._single(
resources.YaraRulesetFavorite.update(
self, id=ruleset_id, favorite=favorite, community=self.community
)
)
def tag_link_get(self, sha256):
"""
Fetch the Tags and Families associated with the given sha256.
:param sha256: The sha256 of the artifact.
:return: A TagLink resource
"""