mirror of
https://gitflic.ru/project/openide/openide.git
synced 2026-09-27 10:03:11 +07:00
maximpopov/84018-implement-dap-pydevd-commands
[debugger] PY-84601 enable "Viea as..." actions in debugpy's variable view Merge-request: IJ-MR-177497 Merged-by: Maxim Popov <maxim.popov@jetbrains.com> GitOrigin-RevId: 86d793bf40fba98335afcd68dbe9ef774aff3c0e
This commit is contained in:
committed by
intellij-monorepo-bot
parent
1426757f14
commit
f3161c5a38
+5
-3
@@ -96,11 +96,11 @@ def load_schema_data():
|
||||
return json_schema_data
|
||||
|
||||
|
||||
def load_custom_schema_data():
|
||||
def load_custom_schema_data(filename_in_current_dir):
|
||||
import os.path
|
||||
import json
|
||||
|
||||
json_file = os.path.join(os.path.dirname(__file__), "debugProtocolCustom.json")
|
||||
json_file = os.path.join(os.path.dirname(__file__), filename_in_current_dir)
|
||||
|
||||
with open(json_file, "rb") as json_contents:
|
||||
json_schema_data = json.loads(json_contents.read())
|
||||
@@ -542,7 +542,9 @@ def gen_debugger_protocol():
|
||||
raise AssertionError("Must be run with Python 3.6 onwards (to keep dict order).")
|
||||
|
||||
classes_to_generate = create_classes_to_generate_structure(load_schema_data())
|
||||
classes_to_generate.update(create_classes_to_generate_structure(load_custom_schema_data()))
|
||||
classes_to_generate.update(create_classes_to_generate_structure(load_custom_schema_data("debugProtocolCustom.json")))
|
||||
classes_to_generate.update(create_classes_to_generate_structure(load_custom_schema_data("debugProtocolCustomPyCharm.json")))
|
||||
|
||||
|
||||
class_to_generate = fill_properties_and_required_from_base(classes_to_generate)
|
||||
|
||||
|
||||
+242
@@ -0,0 +1,242 @@
|
||||
{
|
||||
"$schema": "http://json-schema.org/draft-04/schema#",
|
||||
"title": "Custom Debug Adapter Protocol",
|
||||
"description": "Extension to the DAP to support additional features.",
|
||||
"type": "object",
|
||||
"definitions": {
|
||||
"GetTableRequest": {
|
||||
"allOf": [
|
||||
{
|
||||
"$ref": "#/definitions/Request"
|
||||
},
|
||||
{
|
||||
"type": "object",
|
||||
"description": "Retrieve tabular data (e.g., DataFrame, numpy array, polars) from a variable/expression in the debuggee. Proxies to pydevd InternalTableCommand.",
|
||||
"properties": {
|
||||
"command": {
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"getTable"
|
||||
]
|
||||
},
|
||||
"arguments": {
|
||||
"$ref": "#/definitions/GetTableArguments"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"command",
|
||||
"arguments"
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
"GetTableArguments": {
|
||||
"type": "object",
|
||||
"description": [
|
||||
"Arguments for 'getTable' request.",
|
||||
"Values are evaluated in the context of the given thread/frame.",
|
||||
"The 'command' (initCommand in server) is a Python expression that evaluates to a supported table-like object.",
|
||||
"The 'commandType' selects the operation (DF_INFO, DF_DESCRIBE, VISUALIZATION_DATA, SLICE, SLICE_CSV)."
|
||||
],
|
||||
"properties": {
|
||||
"threadId": {
|
||||
"type": [
|
||||
"string",
|
||||
"integer"
|
||||
],
|
||||
"description": "Thread identifier where the frame/expression should be evaluated."
|
||||
},
|
||||
"frameId": {
|
||||
"type": [
|
||||
"string",
|
||||
"integer"
|
||||
],
|
||||
"description": "Frame identifier within the given thread."
|
||||
},
|
||||
"command": {
|
||||
"type": "string",
|
||||
"description": "Python expression that evaluates to the table-like object (e.g., variable name or expression)."
|
||||
},
|
||||
"commandType": {
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"DF_INFO",
|
||||
"SLICE",
|
||||
"SLICE_CSV",
|
||||
"DF_DESCRIBE",
|
||||
"VISUALIZATION_DATA",
|
||||
"IMAGE_START_CHUNK_LOAD",
|
||||
"IMAGE_CHUNK_LOAD",
|
||||
"INSPECTIONS"
|
||||
]
|
||||
},
|
||||
"start": {
|
||||
"type": [
|
||||
"integer",
|
||||
"null"
|
||||
],
|
||||
"description": "Optional start row index (inclusive) for slice operations."
|
||||
},
|
||||
"end": {
|
||||
"type": [
|
||||
"integer",
|
||||
"null"
|
||||
],
|
||||
"description": "Optional end row index (exclusive) for slice operations."
|
||||
},
|
||||
"format": {
|
||||
"type": [
|
||||
"string",
|
||||
"null"
|
||||
],
|
||||
"description": "Optional backend-specific format hint (e.g., 'json', 'csv', dtype/precision hints)."
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"threadId",
|
||||
"frameId",
|
||||
"commandType"
|
||||
]
|
||||
},
|
||||
"GetTableResponse": {
|
||||
"allOf": [
|
||||
{
|
||||
"$ref": "#/definitions/Response"
|
||||
},
|
||||
{
|
||||
"type": "object",
|
||||
"description": "Response to 'getTable' request.",
|
||||
"properties": {
|
||||
"command": {
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"getTable"
|
||||
]
|
||||
},
|
||||
"body": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"result": {
|
||||
"type": "string",
|
||||
"description": "Opaque string payload with the result of the getTable operation."
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"body"
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
"GetArrayRequest": {
|
||||
"allOf": [
|
||||
{
|
||||
"$ref": "#/definitions/Request"
|
||||
},
|
||||
{
|
||||
"type": "object",
|
||||
"description": "Retrieve array data from a variable/expression in the debuggee. Proxies to pydevd InternalArrayCommand.",
|
||||
"properties": {
|
||||
"command": {
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"getArray"
|
||||
]
|
||||
},
|
||||
"arguments": {
|
||||
"$ref": "#/definitions/GetArrayArguments"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"command",
|
||||
"arguments"
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
"GetArrayArguments": {
|
||||
"type": "object",
|
||||
"description": "Arguments for 'getArray' request.",
|
||||
"properties": {
|
||||
"threadId": {
|
||||
"type": [
|
||||
"string",
|
||||
"integer"
|
||||
],
|
||||
"description": "Thread identifier where the frame/expression should be evaluated."
|
||||
},
|
||||
"frameId": {
|
||||
"type": [
|
||||
"string",
|
||||
"integer"
|
||||
],
|
||||
"description": "Frame identifier within the given thread."
|
||||
},
|
||||
"rowOffset": {
|
||||
"type": "integer",
|
||||
"description": "Row offset"
|
||||
},
|
||||
"colOffset": {
|
||||
"type": "integer",
|
||||
"description": "Col offset"
|
||||
},
|
||||
"rows": {
|
||||
"type": "integer",
|
||||
"description": "Rows"
|
||||
},
|
||||
"cols": {
|
||||
"type": "integer",
|
||||
"description": "Columns"
|
||||
},
|
||||
"format": {
|
||||
"type": [
|
||||
"string",
|
||||
"null"
|
||||
],
|
||||
"description": "Optional backend-specific format hint (e.g., 'json', 'csv', dtype/precision hints)."
|
||||
},
|
||||
"variableName": {
|
||||
"type": "string",
|
||||
"description": "Array variable name"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"threadId",
|
||||
"frameId",
|
||||
"commandType"
|
||||
]
|
||||
},
|
||||
"GetArrayResponse": {
|
||||
"allOf": [
|
||||
{
|
||||
"$ref": "#/definitions/Response"
|
||||
},
|
||||
{
|
||||
"type": "object",
|
||||
"description": "Response to 'getTable' request.",
|
||||
"properties": {
|
||||
"command": {
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"getArray"
|
||||
]
|
||||
},
|
||||
"body": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"result": {
|
||||
"type": "string",
|
||||
"description": "Opaque string payload with the result of the requested operation."
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"body"
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
+8133
-4892
File diff suppressed because it is too large
Load Diff
+39
@@ -0,0 +1,39 @@
|
||||
# Copyright 2000-2024 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
def get_apply():
|
||||
try:
|
||||
from pydevd_nest_asyncio import apply
|
||||
return apply
|
||||
except:
|
||||
return None
|
||||
|
||||
|
||||
def get_eval_async_expression_in_context():
|
||||
try:
|
||||
from pydevd_asyncio_utils import eval_async_expression_in_context
|
||||
return eval_async_expression_in_context
|
||||
except:
|
||||
return None
|
||||
|
||||
|
||||
def get_eval_async_expression():
|
||||
try:
|
||||
from pydevd_asyncio_utils import eval_async_expression
|
||||
return eval_async_expression
|
||||
except:
|
||||
return None
|
||||
|
||||
|
||||
def get_exec_async_code():
|
||||
try:
|
||||
from pydevd_asyncio_utils import exec_async_code
|
||||
return exec_async_code
|
||||
except:
|
||||
return None
|
||||
|
||||
|
||||
def get_asyncio_command_compiler():
|
||||
try:
|
||||
from pydevd_asyncio_utils import asyncio_command_compiler
|
||||
return asyncio_command_compiler
|
||||
except:
|
||||
return None
|
||||
@@ -0,0 +1,8 @@
|
||||
try:
|
||||
xrange = xrange
|
||||
except:
|
||||
# Python 3k does not have it
|
||||
xrange = range
|
||||
|
||||
NUMPY_NUMERIC_TYPES = "biufc"
|
||||
NUMPY_FLOATING_POINT_TYPES = "fc"
|
||||
@@ -0,0 +1,17 @@
|
||||
def get_custom_frame(thread_id, frame_id):
|
||||
'''
|
||||
:param thread_id: This should actually be the frame_id which is returned by add_custom_frame.
|
||||
:param frame_id: This is the actual id() of the frame
|
||||
'''
|
||||
|
||||
CustomFramesContainer.custom_frames_lock.acquire()
|
||||
try:
|
||||
frame_id = int(frame_id)
|
||||
f = CustomFramesContainer.custom_frames[thread_id].frame
|
||||
while f is not None:
|
||||
if id(f) == frame_id:
|
||||
return f
|
||||
f = f.f_back
|
||||
finally:
|
||||
f = None
|
||||
CustomFramesContainer.custom_frames_lock.release()
|
||||
@@ -0,0 +1,149 @@
|
||||
# Copyright 2000-2025 JetBrains s.r.o. Use of this source code is governed by the Apache 2.0 license that can be found in the LICENSE file.
|
||||
|
||||
|
||||
from _pydev_bundle import pydev_log
|
||||
from _pydevd_bundle import pydevd_vars
|
||||
from _pydevd_bundle.pydevd_constants import NEXT_VALUE_SEPARATOR
|
||||
from _pydevd_bundle.pydevd_xml import ExceptionOnEvaluate
|
||||
from _pydevd_bundle.custom.tables.images.pydevd_image_loader import load_image_chunk
|
||||
|
||||
|
||||
class TableCommandType:
|
||||
DF_INFO = "DF_INFO"
|
||||
SLICE = "SLICE"
|
||||
SLICE_CSV = "SLICE_CSV"
|
||||
DESCRIBE = "DF_DESCRIBE"
|
||||
VISUALIZATION_DATA = "VISUALIZATION_DATA"
|
||||
IMAGE_START_CHUNK_LOAD = "IMAGE_START_CHUNK_LOAD"
|
||||
IMAGE_CHUNK_LOAD = "IMAGE_CHUNK_LOAD"
|
||||
|
||||
|
||||
def is_error_on_eval(val):
|
||||
try:
|
||||
# This should be faster than isinstance (but we have to protect against not
|
||||
# having a '__class__' attribute).
|
||||
is_exception_on_eval = val.__class__ == ExceptionOnEvaluate
|
||||
except:
|
||||
is_exception_on_eval = False
|
||||
return is_exception_on_eval
|
||||
|
||||
def exec_image_table_command(init_command, command_type, offset, image_id, f_globals, f_locals):
|
||||
table = pydevd_vars.eval_in_context(init_command, f_globals, f_locals)
|
||||
is_exception_on_eval = is_error_on_eval(table)
|
||||
if is_exception_on_eval:
|
||||
return False, table.result
|
||||
|
||||
image_provider = __get_image_provider(table)
|
||||
if not image_provider:
|
||||
raise RuntimeError('No image provider for: {}'.format(type(table)))
|
||||
|
||||
if command_type == TableCommandType.IMAGE_START_CHUNK_LOAD:
|
||||
return True, image_provider.create_image(table)
|
||||
|
||||
return True, load_image_chunk(offset, image_id)
|
||||
|
||||
|
||||
def exec_table_command(init_command, command_type, start_index, end_index, format, f_globals,
|
||||
f_locals):
|
||||
|
||||
table = pydevd_vars.eval_in_context(init_command, f_globals, f_locals)
|
||||
is_exception_on_eval = is_error_on_eval(table)
|
||||
if is_exception_on_eval:
|
||||
return False, table.result
|
||||
|
||||
table_provider = __get_table_provider(table)
|
||||
if not table_provider:
|
||||
raise RuntimeError('No table data provider for: {}'.format(type(table)))
|
||||
|
||||
res = []
|
||||
if command_type == TableCommandType.DF_INFO:
|
||||
res.append(table_provider.get_type(table))
|
||||
res.append(NEXT_VALUE_SEPARATOR)
|
||||
res.append(table_provider.get_shape(table))
|
||||
res.append(NEXT_VALUE_SEPARATOR)
|
||||
res.append(table_provider.get_head(table))
|
||||
res.append(NEXT_VALUE_SEPARATOR)
|
||||
res.append(table_provider.get_column_types(table))
|
||||
|
||||
elif command_type == TableCommandType.DESCRIBE:
|
||||
res.append(table_provider.get_column_descriptions(table))
|
||||
|
||||
elif command_type == TableCommandType.VISUALIZATION_DATA:
|
||||
res.append(table_provider.get_value_occurrences_count(table))
|
||||
res.append(NEXT_VALUE_SEPARATOR)
|
||||
|
||||
elif command_type == TableCommandType.SLICE:
|
||||
res.append(table_provider.get_data(table, False, start_index, end_index, format))
|
||||
elif command_type == TableCommandType.SLICE_CSV:
|
||||
res.append(table_provider.get_data(table, True, start_index, end_index, format))
|
||||
|
||||
return True, ''.join(res)
|
||||
|
||||
|
||||
def __get_type_name(table):
|
||||
table_data_type = type(table)
|
||||
table_data_type_name = '{}.{}'.format(table_data_type.__module__, table_data_type.__name__)
|
||||
return table_data_type_name
|
||||
|
||||
|
||||
# noinspection PyUnresolvedReferences
|
||||
def __get_table_provider(output):
|
||||
# type: (str) -> Any
|
||||
type_qualified_name = __get_type_name(output)
|
||||
numpy_based_type_qualified_names = ['tensorflow.python.framework.ops.EagerTensor',
|
||||
'tensorflow.python.ops.resource_variable_ops.ResourceVariable',
|
||||
'tensorflow.python.framework.sparse_tensor.SparseTensor',
|
||||
'torch.Tensor']
|
||||
table_provider = None
|
||||
if type_qualified_name in ['pandas.core.frame.DataFrame',
|
||||
'pandas.core.series.Series',
|
||||
'geopandas.geoseries.GeoSeries',
|
||||
'geopandas.geodataframe.GeoDataFrame',
|
||||
'pandera.typing.pandas.DataFrame']:
|
||||
import _pydevd_bundle.custom.tables.pydevd_pandas as table_provider
|
||||
# dict is needed for sort commands
|
||||
elif type_qualified_name == 'builtins.dict':
|
||||
table_type_name = __get_type_name(output['data'])
|
||||
if table_type_name in numpy_based_type_qualified_names:
|
||||
import _pydevd_bundle.custom.tables.pydevd_numpy_based as table_provider
|
||||
else:
|
||||
import _pydevd_bundle.custom.tables.pydevd_numpy as table_provider
|
||||
elif type_qualified_name == 'numpy.ndarray' or type_qualified_name == 'numpy.rec.recarray':
|
||||
import _pydevd_bundle.custom.tables.pydevd_numpy as table_provider
|
||||
elif type_qualified_name in numpy_based_type_qualified_names:
|
||||
import _pydevd_bundle.custom.tables.pydevd_numpy_based as table_provider
|
||||
elif type_qualified_name.startswith('polars') and (
|
||||
type_qualified_name.endswith('DataFrame')
|
||||
or type_qualified_name.endswith('Series')):
|
||||
import _pydevd_bundle.custom.tables.pydevd_polars as table_provider
|
||||
elif type_qualified_name == 'datasets.arrow_dataset.Dataset':
|
||||
import _pydevd_bundle.custom.tables.pydevd_dataset as table_provider
|
||||
|
||||
return table_provider
|
||||
|
||||
|
||||
# noinspection PyUnresolvedReferences
|
||||
def __get_image_provider(output):
|
||||
# type: (str) -> Any
|
||||
type_qualified_name = __get_type_name(output)
|
||||
numpy_based_type_qualified_names = ['tensorflow.python.framework.ops.EagerTensor',
|
||||
'tensorflow.python.ops.resource_variable_ops.ResourceVariable',
|
||||
'tensorflow.python.framework.sparse_tensor.SparseTensor',
|
||||
'torch.Tensor']
|
||||
image_provider = None
|
||||
if type_qualified_name == 'builtins.dict':
|
||||
table_type_name = __get_type_name(output['data'])
|
||||
if table_type_name in numpy_based_type_qualified_names:
|
||||
import _pydevd_bundle.custom.tables.images.pydevd_numpy_based_image as image_provider
|
||||
else:
|
||||
import _pydevd_bundle.custom.tables.images.pydevd_numpy_image as image_provider
|
||||
elif type_qualified_name in numpy_based_type_qualified_names:
|
||||
import _pydevd_bundle.custom.tables.images.pydevd_numpy_based_image as image_provider
|
||||
elif type_qualified_name == 'numpy.ndarray':
|
||||
import _pydevd_bundle.custom.tables.images.pydevd_numpy_image as image_provider
|
||||
elif type_qualified_name in ['PIL.Image.Image', 'PIL.PngImagePlugin.PngImageFile', 'PIL.JpegImagePlugin.JpegImageFile']:
|
||||
import _pydevd_bundle.custom.tables.images.pydevd_pillow_image as image_provider
|
||||
elif type_qualified_name in ['matplotlib.figure.Figure', 'plotly.graph_objs._figure.Figure']:
|
||||
import _pydevd_bundle.custom.tables.images.pydevd_matplotlib_image as image_provider
|
||||
|
||||
return image_provider
|
||||
+58
@@ -0,0 +1,58 @@
|
||||
# Copyright 2000-2021 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license that can be found in the LICENSE file.
|
||||
import inspect
|
||||
|
||||
from _pydevd_bundle import pydevd_utils
|
||||
|
||||
|
||||
class TypeRenderersConstants:
|
||||
new_line = "@_@NEW_LINE_CHAR@_@"
|
||||
tab = "@_@TAB_CHAR@_@"
|
||||
|
||||
|
||||
def try_get_type_renderer_for_var(var, renderers_dict):
|
||||
try:
|
||||
cls = var.__class__
|
||||
cls_name = cls.__name__
|
||||
renderers_for_name = renderers_dict.get(cls_name)
|
||||
if renderers_for_name is None:
|
||||
return None
|
||||
|
||||
module_name = cls.__module__
|
||||
qualified_name = module_name + "." + cls_name
|
||||
|
||||
# for builtins
|
||||
builtin_module = int.__module__
|
||||
if module_name == builtin_module:
|
||||
for render in renderers_for_name:
|
||||
if render.type_canonical_import_path == qualified_name:
|
||||
return render
|
||||
|
||||
# for classes which defined in project directory
|
||||
try:
|
||||
src_file = inspect.getfile(cls)
|
||||
except:
|
||||
src_file = None
|
||||
|
||||
if src_file is not None and pydevd_utils.in_project_roots(src_file):
|
||||
for render in renderers_for_name:
|
||||
if render.type_src_file == src_file:
|
||||
return render
|
||||
return None
|
||||
|
||||
# by qualified name
|
||||
for render in renderers_for_name:
|
||||
if render.type_qualified_name == qualified_name:
|
||||
return render
|
||||
|
||||
# by module root and class name
|
||||
# (if module contains only one class with the same name)
|
||||
module_root = module_name.split(".")[0]
|
||||
for render in renderers_for_name:
|
||||
if render.module_root_has_one_type_with_same_name:
|
||||
renderer_module_root = render.type_canonical_import_path.split(".")[0]
|
||||
if renderer_module_root == module_root:
|
||||
return render
|
||||
except:
|
||||
pass
|
||||
|
||||
return None
|
||||
@@ -0,0 +1,11 @@
|
||||
class VariableWithOffset(object):
|
||||
def __init__(self, data, offset):
|
||||
self.data, self.offset = data, offset
|
||||
|
||||
|
||||
def eval_expression(expression, globals, locals):
|
||||
eval_func = get_eval_async_expression_in_context()
|
||||
if eval_func is not None:
|
||||
return eval_func(expression, globals, locals, False)
|
||||
|
||||
return eval(expression, globals, locals)
|
||||
@@ -0,0 +1,888 @@
|
||||
""" pydevd_vars deals with variables:
|
||||
resolution/conversion to XML.
|
||||
"""
|
||||
import math
|
||||
import pickle
|
||||
|
||||
from _pydev_bundle.pydev_imports import quote
|
||||
from _pydev_bundle._pydev_saved_modules import thread, threading
|
||||
from _pydevd_bundle.pydevd_constants import get_frame, get_current_thread_id
|
||||
from _pydevd_bundle.custom.pydevd_constants import xrange, NUMPY_NUMERIC_TYPES, NUMPY_FLOATING_POINT_TYPES
|
||||
from _pydevd_bundle.custom.pydevd_custom_frames import get_custom_frame
|
||||
from _pydevd_bundle.custom.pydevd_user_type_renderers_utils import try_get_type_renderer_for_var
|
||||
from _pydevd_bundle.pydevd_xml import ExceptionOnEvaluate, get_type, var_to_xml
|
||||
from _pydevd_bundle.custom.pydevd_asyncio_provider import get_eval_async_expression
|
||||
|
||||
try:
|
||||
from StringIO import StringIO
|
||||
except ImportError:
|
||||
from io import StringIO
|
||||
import sys # @Reimport
|
||||
|
||||
try:
|
||||
from collections import OrderedDict
|
||||
except:
|
||||
OrderedDict = dict
|
||||
|
||||
from _pydev_bundle._pydev_saved_modules import threading
|
||||
import traceback
|
||||
from _pydevd_bundle import pydevd_save_locals
|
||||
from _pydev_bundle.pydev_imports import Exec, execfile
|
||||
from _pydevd_bundle.custom.pydevd_utils import VariableWithOffset, eval_expression
|
||||
from _pydevd_bundle.pydevd_constants import IS_PY313_OR_GREATER
|
||||
|
||||
SENTINEL_VALUE = []
|
||||
DEFAULT_DF_FORMAT = "s"
|
||||
|
||||
# ------------------------------------------------------------------------------------------------------ class for errors
|
||||
|
||||
class VariableError(RuntimeError): pass
|
||||
|
||||
|
||||
class FrameNotFoundError(RuntimeError): pass
|
||||
|
||||
|
||||
def _iter_frames(initialFrame):
|
||||
'''NO-YIELD VERSION: Iterates through all the frames starting at the specified frame (which will be the first returned item)'''
|
||||
# cannot use yield
|
||||
frames = []
|
||||
|
||||
while initialFrame is not None:
|
||||
frames.append(initialFrame)
|
||||
initialFrame = initialFrame.f_back
|
||||
|
||||
return frames
|
||||
|
||||
|
||||
def dump_frames(thread_id):
|
||||
sys.stdout.write('dumping frames\n')
|
||||
if thread_id != get_current_thread_id(threading.current_thread()):
|
||||
raise VariableError("find_frame: must execute on same thread")
|
||||
|
||||
curFrame = get_frame()
|
||||
for frame in _iter_frames(curFrame):
|
||||
sys.stdout.write('%s\n' % pickle.dumps(frame))
|
||||
|
||||
|
||||
# ===============================================================================
|
||||
# AdditionalFramesContainer
|
||||
# ===============================================================================
|
||||
class AdditionalFramesContainer:
|
||||
lock = thread.allocate_lock()
|
||||
additional_frames = {} # dict of dicts
|
||||
|
||||
|
||||
def add_additional_frame_by_id(thread_id, frames_by_id):
|
||||
AdditionalFramesContainer.additional_frames[thread_id] = frames_by_id
|
||||
|
||||
|
||||
addAdditionalFrameById = add_additional_frame_by_id # Backward compatibility
|
||||
|
||||
|
||||
def remove_additional_frame_by_id(thread_id):
|
||||
del AdditionalFramesContainer.additional_frames[thread_id]
|
||||
|
||||
|
||||
removeAdditionalFrameById = remove_additional_frame_by_id # Backward compatibility
|
||||
|
||||
|
||||
def has_additional_frames_by_id(thread_id):
|
||||
return thread_id in AdditionalFramesContainer.additional_frames
|
||||
|
||||
|
||||
def get_additional_frames_by_id(thread_id):
|
||||
return AdditionalFramesContainer.additional_frames.get(thread_id)
|
||||
|
||||
|
||||
def find_frame(thread_id, frame_id):
|
||||
""" returns a frame on the thread that has a given frame_id """
|
||||
try:
|
||||
curr_thread_id = get_current_thread_id(threading.current_thread())
|
||||
if thread_id != curr_thread_id:
|
||||
try:
|
||||
return get_custom_frame(thread_id, frame_id) # I.e.: thread_id could be a stackless frame id + thread_id.
|
||||
except:
|
||||
pass
|
||||
|
||||
raise VariableError("find_frame: must execute on same thread (%s != %s)" % (thread_id, curr_thread_id))
|
||||
|
||||
lookingFor = int(frame_id)
|
||||
|
||||
if AdditionalFramesContainer.additional_frames:
|
||||
if thread_id in AdditionalFramesContainer.additional_frames:
|
||||
frame = AdditionalFramesContainer.additional_frames[thread_id].get(lookingFor)
|
||||
|
||||
if frame is not None:
|
||||
return frame
|
||||
|
||||
curFrame = get_frame()
|
||||
if frame_id == "*":
|
||||
return curFrame # any frame is specified with "*"
|
||||
|
||||
frameFound = None
|
||||
|
||||
for frame in _iter_frames(curFrame):
|
||||
if lookingFor == id(frame):
|
||||
frameFound = frame
|
||||
del frame
|
||||
break
|
||||
|
||||
del frame
|
||||
|
||||
# Important: python can hold a reference to the frame from the current context
|
||||
# if an exception is raised, so, if we don't explicitly add those deletes
|
||||
# we might have those variables living much more than we'd want to.
|
||||
|
||||
# I.e.: sys.exc_info holding reference to frame that raises exception (so, other places
|
||||
# need to call sys.exc_clear())
|
||||
del curFrame
|
||||
|
||||
if frameFound is None:
|
||||
msgFrames = ''
|
||||
i = 0
|
||||
|
||||
for frame in _iter_frames(get_frame()):
|
||||
i += 1
|
||||
msgFrames += str(id(frame))
|
||||
if i % 5 == 0:
|
||||
msgFrames += '\n'
|
||||
else:
|
||||
msgFrames += ' - '
|
||||
|
||||
# Note: commented this error message out (it may commonly happen
|
||||
# if a message asking for a frame is issued while a thread is paused
|
||||
# but the thread starts running before the message is actually
|
||||
# handled).
|
||||
# Leaving code to uncomment during tests.
|
||||
# err_msg = '''find_frame: frame not found.
|
||||
# Looking for thread_id:%s, frame_id:%s
|
||||
# Current thread_id:%s, available frames:
|
||||
# %s\n
|
||||
# ''' % (thread_id, lookingFor, curr_thread_id, msgFrames)
|
||||
#
|
||||
# sys.stderr.write(err_msg)
|
||||
return None
|
||||
|
||||
return frameFound
|
||||
except:
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
return None
|
||||
|
||||
|
||||
def getVariable(thread_id, frame_id, scope, attrs):
|
||||
"""
|
||||
returns the value of a variable
|
||||
|
||||
:scope: can be BY_ID, EXPRESSION, GLOBAL, LOCAL, FRAME
|
||||
|
||||
BY_ID means we'll traverse the list of all objects alive to get the object.
|
||||
|
||||
:attrs: after reaching the proper scope, we have to get the attributes until we find
|
||||
the proper location (i.e.: obj\tattr1\tattr2).
|
||||
|
||||
:note: when BY_ID is used, the frame_id is considered the id of the object to find and
|
||||
not the frame (as we don't care about the frame in this case).
|
||||
"""
|
||||
if scope == 'BY_ID':
|
||||
if thread_id != get_current_thread_id(threading.current_thread()):
|
||||
raise VariableError("getVariable: must execute on same thread")
|
||||
|
||||
try:
|
||||
import gc
|
||||
objects = gc.get_objects()
|
||||
except:
|
||||
pass # Not all python variants have it.
|
||||
else:
|
||||
frame_id = int(frame_id)
|
||||
for var in objects:
|
||||
if id(var) == frame_id:
|
||||
if attrs is not None:
|
||||
attrList = attrs.split('\t')
|
||||
for k in attrList:
|
||||
_type, _typeName, resolver = get_type(var)
|
||||
var = resolver.resolve(var, k)
|
||||
|
||||
return var
|
||||
|
||||
# If it didn't return previously, we coudn't find it by id (i.e.: alrceady garbage collected).
|
||||
sys.stderr.write('Unable to find object with id: %s\n' % (frame_id,))
|
||||
return None
|
||||
|
||||
frame = find_frame(thread_id, frame_id)
|
||||
if frame is None:
|
||||
return {}
|
||||
|
||||
if attrs is not None:
|
||||
attrList = attrs.split('\t')
|
||||
else:
|
||||
attrList = []
|
||||
|
||||
for attr in attrList:
|
||||
attr.replace("@_@TAB_CHAR@_@", '\t')
|
||||
|
||||
if scope == 'EXPRESSION':
|
||||
for count in xrange(len(attrList)):
|
||||
if count == 0:
|
||||
# An Expression can be in any scope (globals/locals), therefore it needs to evaluated as an expression
|
||||
var = evaluate_expression(thread_id, frame_id, attrList[count], False)
|
||||
else:
|
||||
_type, _typeName, resolver = get_type(var)
|
||||
var = resolver.resolve(var, attrList[count])
|
||||
else:
|
||||
if scope == "GLOBAL":
|
||||
var = frame.f_globals
|
||||
del attrList[0] # globals are special, and they get a single dummy unused attribute
|
||||
else:
|
||||
# in a frame access both locals and globals as Python does
|
||||
var = {}
|
||||
var.update(frame.f_globals)
|
||||
var.update(frame.f_locals)
|
||||
|
||||
for k in attrList:
|
||||
_type, _typeName, resolver = get_type(var)
|
||||
var = resolver.resolve(var, k)
|
||||
|
||||
return var
|
||||
|
||||
|
||||
def get_offset(attrs):
|
||||
"""
|
||||
Extract offset from the given attributes.
|
||||
|
||||
:param attrs: The string of a compound variable fields split by tabs.
|
||||
If an offset is given, it must go the first element.
|
||||
:return: The value of offset if given or 0.
|
||||
"""
|
||||
offset = 0
|
||||
if attrs is not None:
|
||||
try:
|
||||
offset = int(attrs.split('\t')[0])
|
||||
except ValueError:
|
||||
pass
|
||||
return offset
|
||||
|
||||
|
||||
def _resolve_default_variable_fields(var, resolver, offset):
|
||||
return resolver.get_dictionary(VariableWithOffset(var, offset) if offset else var)
|
||||
|
||||
|
||||
def _resolve_custom_variable_fields(var, var_expr, resolver, offset, type_renderer, frame_info=None):
|
||||
val_dict = OrderedDict()
|
||||
if type_renderer.is_default_children or type_renderer.append_default_children:
|
||||
default_val_dict = _resolve_default_variable_fields(var, resolver, offset)
|
||||
if len(val_dict) == 0:
|
||||
return default_val_dict
|
||||
for (name, value) in default_val_dict.items():
|
||||
val_dict[name] = value
|
||||
|
||||
return val_dict
|
||||
|
||||
|
||||
def resolve_compound_variable_fields(thread_id, frame_id, scope, attrs, user_type_renderers={}):
|
||||
"""
|
||||
Resolve compound variable in debugger scopes by its name and attributes
|
||||
|
||||
:param thread_id: id of the variable's thread
|
||||
:param frame_id: id of the variable's frame
|
||||
:param scope: can be BY_ID, EXPRESSION, GLOBAL, LOCAL, FRAME
|
||||
:param attrs: after reaching the proper scope, we have to get the attributes until we find
|
||||
the proper location (i.e.: obj\tattr1\tattr2)
|
||||
:param user_type_renderers: a dictionary with user type renderers
|
||||
:return: a dictionary of variables's fields
|
||||
|
||||
:note: PyCharm supports progressive loading of large collections and uses the `attrs`
|
||||
parameter to pass the offset, e.g. 300\t\\obj\tattr1\tattr2 should return
|
||||
the value of attr2 starting from the 300th element. This hack makes it possible
|
||||
to add the support of progressive loading without extending of the protocol.
|
||||
"""
|
||||
offset = get_offset(attrs)
|
||||
|
||||
orig_attrs, attrs = attrs, attrs.split('\t', 1)[1] if offset else attrs
|
||||
|
||||
var = getVariable(thread_id, frame_id, scope, attrs)
|
||||
|
||||
var_expr = ".".join(attrs.split('\t'))
|
||||
|
||||
try:
|
||||
_type, _typeName, resolver = get_type(var)
|
||||
|
||||
type_renderer = try_get_type_renderer_for_var(var, user_type_renderers)
|
||||
if type_renderer is not None and offset == 0:
|
||||
frame_info = (thread_id, frame_id)
|
||||
return _typeName, _resolve_custom_variable_fields(
|
||||
var, var_expr, resolver, offset, type_renderer, frame_info
|
||||
)
|
||||
|
||||
return _typeName, _resolve_default_variable_fields(var, resolver, offset)
|
||||
|
||||
except:
|
||||
sys.stderr.write('Error evaluating: thread_id: %s\nframe_id: %s\nscope: %s\nattrs: %s\n' % (
|
||||
thread_id, frame_id, scope, orig_attrs,))
|
||||
traceback.print_exc()
|
||||
|
||||
|
||||
def resolve_var_object(var, attrs):
|
||||
"""
|
||||
Resolve variable's attribute
|
||||
|
||||
:param var: an object of variable
|
||||
:param attrs: a sequence of variable's attributes separated by \t (i.e.: obj\tattr1\tattr2)
|
||||
:return: a value of resolved variable's attribute
|
||||
"""
|
||||
if attrs is not None:
|
||||
attr_list = attrs.split('\t')
|
||||
else:
|
||||
attr_list = []
|
||||
for k in attr_list:
|
||||
type, _typeName, resolver = get_type(var)
|
||||
var = resolver.resolve(var, k)
|
||||
return var
|
||||
|
||||
|
||||
def resolve_compound_var_object_fields(var, attrs, user_type_renderers={}):
|
||||
"""
|
||||
Resolve compound variable by its object and attributes
|
||||
|
||||
:param var: an object of variable
|
||||
:param attrs: a sequence of variable's attributes separated by \t (i.e.: obj\tattr1\tattr2)
|
||||
:param user_type_renderers: a dictionary with user type renderers
|
||||
:return: a dictionary of variables's fields
|
||||
"""
|
||||
namespace = var
|
||||
|
||||
offset = get_offset(attrs)
|
||||
|
||||
attrs = attrs.split('\t', 1)[1] if offset else attrs
|
||||
|
||||
attr_list = attrs.split('\t')
|
||||
|
||||
var_expr = ".".join(attr_list)
|
||||
|
||||
for k in attr_list:
|
||||
type, _typeName, resolver = get_type(var)
|
||||
var = resolver.resolve(var, k)
|
||||
|
||||
try:
|
||||
type, _typeName, resolver = get_type(var)
|
||||
|
||||
type_renderer = try_get_type_renderer_for_var(var, user_type_renderers)
|
||||
if type_renderer is not None and offset == 0:
|
||||
return _resolve_custom_variable_fields(
|
||||
var, var_expr, resolver, offset, type_renderer
|
||||
)
|
||||
|
||||
return _resolve_default_variable_fields(var, resolver, offset)
|
||||
except:
|
||||
traceback.print_exc()
|
||||
|
||||
|
||||
def custom_operation(thread_id, frame_id, scope, attrs, style, code_or_file, operation_fn_name):
|
||||
"""
|
||||
We'll execute the code_or_file and then search in the namespace the operation_fn_name to execute with the given var.
|
||||
|
||||
code_or_file: either some code (i.e.: from pprint import pprint) or a file to be executed.
|
||||
operation_fn_name: the name of the operation to execute after the exec (i.e.: pprint)
|
||||
"""
|
||||
expressionValue = getVariable(thread_id, frame_id, scope, attrs)
|
||||
|
||||
try:
|
||||
namespace = {'__name__': '<custom_operation>'}
|
||||
if style == "EXECFILE":
|
||||
namespace['__file__'] = code_or_file
|
||||
execfile(code_or_file, namespace, namespace)
|
||||
else: # style == EXEC
|
||||
namespace['__file__'] = '<customOperationCode>'
|
||||
Exec(code_or_file, namespace, namespace)
|
||||
|
||||
return str(namespace[operation_fn_name](expressionValue))
|
||||
except:
|
||||
traceback.print_exc()
|
||||
|
||||
|
||||
def get_eval_exception_msg(expression, locals):
|
||||
s = StringIO()
|
||||
traceback.print_exc(file=s)
|
||||
result = s.getvalue()
|
||||
|
||||
try:
|
||||
try:
|
||||
etype, value, tb = sys.exc_info()
|
||||
result = value
|
||||
finally:
|
||||
etype = value = tb = None
|
||||
except:
|
||||
pass
|
||||
|
||||
result = ExceptionOnEvaluate(result)
|
||||
|
||||
# Ok, we have the initial error message, but let's see if we're dealing with a name mangling error...
|
||||
try:
|
||||
if '__' in expression:
|
||||
# Try to handle '__' name mangling...
|
||||
split = expression.split('.')
|
||||
curr = locals.get(split[0])
|
||||
for entry in split[1:]:
|
||||
if entry.startswith('__') and not hasattr(curr, entry):
|
||||
entry = '_%s%s' % (curr.__class__.__name__, entry)
|
||||
curr = getattr(curr, entry)
|
||||
|
||||
result = curr
|
||||
except:
|
||||
pass
|
||||
return result
|
||||
|
||||
|
||||
def eval_in_context(expression, globals, locals):
|
||||
try:
|
||||
result = eval_expression(expression, globals, locals)
|
||||
except Exception:
|
||||
result = get_eval_exception_msg(expression, locals)
|
||||
return result
|
||||
|
||||
|
||||
def evaluate_expression(thread_id, frame_id, expression, doExec):
|
||||
'''returns the result of the evaluated expression
|
||||
@param doExec: determines if we should do an exec or an eval
|
||||
'''
|
||||
frame = find_frame(thread_id, frame_id)
|
||||
if frame is None:
|
||||
return
|
||||
|
||||
# Not using frame.f_globals because of https://sourceforge.net/tracker2/?func=detail&aid=2541355&group_id=85796&atid=577329
|
||||
# (Names not resolved in generator expression in method)
|
||||
# See message: http://mail.python.org/pipermail/python-list/2009-January/526522.html
|
||||
updated_globals = {}
|
||||
updated_globals.update(frame.f_globals)
|
||||
updated_globals.update(frame.f_locals) # locals later because it has precedence over the actual globals
|
||||
|
||||
try:
|
||||
expression = str(expression.replace('@LINE@', '\n'))
|
||||
eval_func = get_eval_async_expression()
|
||||
if eval_func is not None:
|
||||
return eval_func(expression, updated_globals, frame, doExec, get_eval_exception_msg)
|
||||
|
||||
if doExec:
|
||||
try:
|
||||
# try to make it an eval (if it is an eval we can print it, otherwise we'll exec it and
|
||||
# it will have whatever the user actually did)
|
||||
compiled = compile(expression, '<string>', 'eval')
|
||||
except:
|
||||
Exec(expression, updated_globals, frame.f_locals)
|
||||
pydevd_save_locals.save_locals(frame)
|
||||
else:
|
||||
result = eval(compiled, updated_globals, frame.f_locals)
|
||||
if result is not None: # Only print if it's not None (as python does)
|
||||
sys.stdout.write('%s\n' % (result,))
|
||||
return
|
||||
|
||||
else:
|
||||
return eval_in_context(expression, updated_globals, frame.f_locals)
|
||||
finally:
|
||||
# Should not be kept alive if an exception happens and this frame is kept in the stack.
|
||||
del updated_globals
|
||||
del frame
|
||||
|
||||
|
||||
def change_attr_expression(thread_id, frame_id, attr, expression, dbg, value=SENTINEL_VALUE):
|
||||
'''Changes some attribute in a given frame.
|
||||
'''
|
||||
frame = find_frame(thread_id, frame_id)
|
||||
if frame is None:
|
||||
return
|
||||
|
||||
try:
|
||||
expression = expression.replace('@LINE@', '\n')
|
||||
|
||||
if dbg.plugin and value is SENTINEL_VALUE:
|
||||
result = dbg.plugin.change_variable(frame, attr, expression)
|
||||
if result:
|
||||
return result
|
||||
|
||||
if value is SENTINEL_VALUE:
|
||||
# It is possible to have variables with names like '.0', ',,,foo', etc in scope by setting them with
|
||||
# `sys._getframe().f_locals`. In particular, the '.0' variable name is used to denote the list iterator when we stop in
|
||||
# list comprehension expressions. This variable evaluates to 0. by `eval`, which is not what we want and this is the main
|
||||
# reason we have to check if the expression exists in the global and local scopes before trying to evaluate it.
|
||||
value = frame.f_locals.get(expression) or frame.f_globals.get(expression) or eval(expression, frame.f_globals, frame.f_locals)
|
||||
|
||||
if attr[:7] == "Globals":
|
||||
attr = attr[8:]
|
||||
if is_complex(attr):
|
||||
Exec('%s=%s' % (attr, expression), frame.f_globals, frame.f_globals)
|
||||
return value
|
||||
if attr in frame.f_globals:
|
||||
frame.f_globals[attr] = value
|
||||
return frame.f_globals[attr]
|
||||
else:
|
||||
if pydevd_save_locals.is_save_locals_available():
|
||||
if is_complex(attr):
|
||||
if IS_PY313_OR_GREATER:
|
||||
Exec('%s=%s' % (attr, expression), frame.f_globals, frame.f_locals)
|
||||
else:
|
||||
Exec('%s=%s' % (attr, expression), frame.f_locals, frame.f_locals)
|
||||
return value
|
||||
frame.f_locals[attr] = value
|
||||
pydevd_save_locals.save_locals(frame)
|
||||
return frame.f_locals[attr]
|
||||
|
||||
# default way (only works for changing it in the topmost frame)
|
||||
result = value
|
||||
Exec('%s=%s' % (attr, expression), frame.f_globals, frame.f_locals)
|
||||
return result
|
||||
|
||||
except Exception:
|
||||
traceback.print_exc()
|
||||
|
||||
def is_complex(attr):
|
||||
complex_indicators = ['[', ']', '.']
|
||||
for indicator in complex_indicators:
|
||||
if attr.find(indicator) != -1:
|
||||
return True
|
||||
return False
|
||||
|
||||
MAXIMUM_ARRAY_SIZE = float('inf')
|
||||
|
||||
|
||||
def array_to_xml(array, name, roffset, coffset, rows, cols, format):
|
||||
array, xml, r, c, f = array_to_meta_xml(array, name, format)
|
||||
format = '%' + f
|
||||
if rows == -1 and cols == -1:
|
||||
rows = r
|
||||
cols = c
|
||||
|
||||
rows = min(rows, MAXIMUM_ARRAY_SIZE)
|
||||
cols = min(cols, MAXIMUM_ARRAY_SIZE)
|
||||
|
||||
if rows == 0 and cols == 0:
|
||||
return xml
|
||||
|
||||
# there is no obvious rule for slicing (at least 5 choices)
|
||||
if len(array) == 1 and (rows > 1 or cols > 1):
|
||||
array = array[0]
|
||||
if array.size > len(array):
|
||||
array = array[roffset:, coffset:]
|
||||
rows = min(rows, len(array))
|
||||
cols = min(cols, len(array[0]))
|
||||
if len(array) == 1:
|
||||
array = array[0]
|
||||
elif array.size == len(array):
|
||||
if roffset == 0 and rows == 1:
|
||||
array = array[coffset:]
|
||||
cols = min(cols, len(array))
|
||||
elif coffset == 0 and cols == 1:
|
||||
array = array[roffset:]
|
||||
rows = min(rows, len(array))
|
||||
|
||||
def get_value(row, col):
|
||||
value = array
|
||||
if rows == 1 or cols == 1:
|
||||
if rows == 1 and cols == 1:
|
||||
value = array[0]
|
||||
else:
|
||||
value = array[(col if rows == 1 else row)]
|
||||
if "ndarray" in str(type(value)):
|
||||
value = value[0]
|
||||
else:
|
||||
value = array[row][col]
|
||||
return value
|
||||
xml += array_data_to_xml(rows, cols, lambda r: (get_value(r, c) for c in range(cols)), format)
|
||||
return xml
|
||||
|
||||
|
||||
def tf_to_xml(tensor, name, roffset, coffset, rows, cols, format):
|
||||
try:
|
||||
return array_to_xml(tensor.numpy(), name, roffset, coffset, rows, cols, format)
|
||||
except TypeError:
|
||||
return array_to_xml(tensor.to_dense().numpy(), name, roffset, coffset, rows, cols, format)
|
||||
|
||||
|
||||
def torch_to_xml(tensor, name, roffset, coffset, rows, cols, format):
|
||||
try:
|
||||
if tensor.requires_grad:
|
||||
tensor = tensor.detach()
|
||||
return array_to_xml(tensor.numpy(), name, roffset, coffset, rows, cols, format)
|
||||
except TypeError:
|
||||
return array_to_xml(tensor.to_dense().numpy(), name, roffset, coffset, rows, cols, format)
|
||||
|
||||
|
||||
def tf_sparse_to_xml(tensor, name, roffset, coffset, rows, cols, format):
|
||||
try:
|
||||
import tensorflow as tf
|
||||
return tf_to_xml(tf.sparse.to_dense(tf.sparse.reorder(tensor)), name, roffset, coffset, rows, cols, format)
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
|
||||
class ExceedingArrayDimensionsException(Exception):
|
||||
pass
|
||||
|
||||
|
||||
def array_to_meta_xml(array, name, format):
|
||||
type = array.dtype.kind
|
||||
slice = name
|
||||
l = len(array.shape)
|
||||
|
||||
if l == 0:
|
||||
rows, cols = 0, 0
|
||||
bounds = (0, 0)
|
||||
return array, slice_to_xml(name, rows, cols, format, "", bounds), rows, cols, format
|
||||
|
||||
try:
|
||||
import numpy as np
|
||||
if isinstance(array, np.recarray) and l > 1:
|
||||
slice = "{}['{}']".format(slice, array.dtype.names[0])
|
||||
array = array[array.dtype.names[0]]
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
# initial load, compute slice
|
||||
if format == '%':
|
||||
if l > 2:
|
||||
slice += '[0]' * (l - 2)
|
||||
for r in range(l - 2):
|
||||
array = array[0]
|
||||
if type == 'f':
|
||||
format = '.5f'
|
||||
elif type == 'i' or type == 'u':
|
||||
format = 'd'
|
||||
else:
|
||||
format = 's'
|
||||
else:
|
||||
format = format.replace('%', '')
|
||||
|
||||
l = len(array.shape)
|
||||
reslice = ""
|
||||
if l > 2:
|
||||
raise ExceedingArrayDimensionsException()
|
||||
elif l == 1:
|
||||
# special case with 1D arrays arr[i, :] - row, but arr[:, i] - column with equal shape and ndim
|
||||
# http://stackoverflow.com/questions/16837946/numpy-a-2-rows-1-column-file-loadtxt-returns-1row-2-columns
|
||||
# explanation: http://stackoverflow.com/questions/15165170/how-do-i-maintain-row-column-orientation-of-vectors-in-numpy?rq=1
|
||||
# we use kind of a hack - get information about memory from C_CONTIGUOUS
|
||||
cols = 1
|
||||
rows = len(array)
|
||||
if rows < len(array):
|
||||
reslice = '[0:%s]' % (rows)
|
||||
array = array[0:rows]
|
||||
elif l == 2:
|
||||
rows = array.shape[-2]
|
||||
cols = array.shape[-1]
|
||||
if cols < array.shape[-1] or rows < array.shape[-2]:
|
||||
reslice = '[0:%s, 0:%s]' % (rows, cols)
|
||||
array = array[0:rows, 0:cols]
|
||||
|
||||
# avoid slice duplication
|
||||
if not slice.endswith(reslice):
|
||||
slice += reslice
|
||||
|
||||
bounds = (0, 0)
|
||||
if type in NUMPY_NUMERIC_TYPES and array.size != 0:
|
||||
bounds = (array.min(), array.max())
|
||||
return array, slice_to_xml(slice, rows, cols, format, type, bounds), rows, cols, format
|
||||
|
||||
|
||||
def get_column_formatter_by_type(initial_format, column_type):
|
||||
if column_type in NUMPY_NUMERIC_TYPES and initial_format:
|
||||
if column_type in NUMPY_FLOATING_POINT_TYPES and initial_format.strip() == DEFAULT_DF_FORMAT:
|
||||
# use custom formatting for floats when default formatting is set
|
||||
return array_default_format(column_type)
|
||||
return initial_format
|
||||
else:
|
||||
return array_default_format(column_type)
|
||||
|
||||
|
||||
def get_formatted_row_elements(row, iat, dim, cols, format, dtypes):
|
||||
for c in range(cols):
|
||||
val = iat[row, c] if dim > 1 else iat[row]
|
||||
col_formatter = get_column_formatter_by_type(format, dtypes[c])
|
||||
try:
|
||||
if val != val:
|
||||
yield "nan"
|
||||
else:
|
||||
yield ("%" + col_formatter) % (val,)
|
||||
except TypeError:
|
||||
yield ("%" + DEFAULT_DF_FORMAT) % (val,)
|
||||
|
||||
|
||||
def array_default_format(type):
|
||||
if type == 'f':
|
||||
return '.5f'
|
||||
elif type == 'i' or type == 'u':
|
||||
return 'd'
|
||||
else:
|
||||
return 's'
|
||||
|
||||
|
||||
def get_label(label):
|
||||
return str(label) if not isinstance(label, tuple) else '/'.join(map(str, label))
|
||||
|
||||
|
||||
DATAFRAME_HEADER_LOAD_MAX_SIZE = 100
|
||||
|
||||
class IAtPolarsAccessor:
|
||||
def __init__(self, ps):
|
||||
self.ps = ps
|
||||
|
||||
def __getitem__(self, row):
|
||||
return self.ps[row]
|
||||
|
||||
|
||||
def dataframe_to_xml(df, name, roffset, coffset, rows, cols, format):
|
||||
"""
|
||||
:type df: pandas.core.frame.DataFrame
|
||||
:type name: str
|
||||
:type coffset: int
|
||||
:type roffset: int
|
||||
:type rows: int
|
||||
:type cols: int
|
||||
:type format: str
|
||||
|
||||
|
||||
"""
|
||||
original_df = df
|
||||
dim = len(df.axes) if hasattr(df, 'axes') else -1
|
||||
num_rows = df.shape[0]
|
||||
num_cols = df.shape[1] if dim > 1 else 1
|
||||
format = format.replace('%', '')
|
||||
|
||||
if not format:
|
||||
if num_rows > 0 and num_cols == 1: # series or data frame with one column
|
||||
try:
|
||||
kind = df.dtype.kind
|
||||
except AttributeError:
|
||||
try:
|
||||
kind = df.dtypes[0].kind
|
||||
except (IndexError, KeyError, AttributeError):
|
||||
kind = 'O'
|
||||
format = array_default_format(kind)
|
||||
else:
|
||||
format = array_default_format(DEFAULT_DF_FORMAT)
|
||||
|
||||
xml = slice_to_xml(name, num_rows, num_cols, format, "", (0, 0))
|
||||
|
||||
if (rows, cols) == (-1, -1):
|
||||
rows, cols = num_rows, num_cols
|
||||
|
||||
elif (rows, cols) == (0, 0):
|
||||
# return header only
|
||||
r = min(num_rows, DATAFRAME_HEADER_LOAD_MAX_SIZE)
|
||||
c = min(num_cols, DATAFRAME_HEADER_LOAD_MAX_SIZE)
|
||||
xml += header_data_to_xml(r, c, [""] * num_cols, [(0, 0)] * num_cols, lambda x: DEFAULT_DF_FORMAT, original_df, dim)
|
||||
return xml
|
||||
|
||||
rows = min(rows, MAXIMUM_ARRAY_SIZE)
|
||||
cols = min(cols, MAXIMUM_ARRAY_SIZE, num_cols)
|
||||
# need to precompute column bounds here before slicing!
|
||||
col_bounds = [None] * cols
|
||||
dtypes = [None] * cols
|
||||
if dim > 1:
|
||||
for col in range(cols):
|
||||
dtype = df.dtypes.iloc[coffset + col].kind
|
||||
dtypes[col] = dtype
|
||||
if dtype in NUMPY_NUMERIC_TYPES and df.size != 0:
|
||||
cvalues = df.iloc[:, coffset + col]
|
||||
bounds = (cvalues.min(), cvalues.max())
|
||||
else:
|
||||
bounds = (0, 0)
|
||||
col_bounds[col] = bounds
|
||||
elif dim == -1:
|
||||
dtype = '0'
|
||||
dtypes[0] = dtype
|
||||
col_bounds[0] = (df.min(), df.max()) if dtype in NUMPY_NUMERIC_TYPES and df.size != 0 else (0, 0)
|
||||
else:
|
||||
dtype = df.dtype.kind
|
||||
dtypes[0] = dtype
|
||||
col_bounds[0] = (df.min(), df.max()) if dtype in NUMPY_NUMERIC_TYPES and df.size != 0 else (0, 0)
|
||||
|
||||
if dim > 1:
|
||||
df = df.iloc[roffset: roffset + rows, coffset: coffset + cols]
|
||||
elif dim == -1:
|
||||
df = df[roffset: roffset + rows]
|
||||
else:
|
||||
df = df.iloc[roffset: roffset + rows]
|
||||
|
||||
rows = df.shape[0]
|
||||
cols = df.shape[1] if dim > 1 else 1
|
||||
|
||||
def col_to_format(column_type):
|
||||
return get_column_formatter_by_type(format, column_type)
|
||||
|
||||
if dim == -1:
|
||||
iat = IAtPolarsAccessor(df)
|
||||
elif dim == 1 or len(df.columns.unique()) == len(df.columns):
|
||||
iat = df.iat
|
||||
else:
|
||||
iat = df.iloc
|
||||
|
||||
def formatted_row_elements(row):
|
||||
return get_formatted_row_elements(row, iat, dim, cols, format, dtypes)
|
||||
|
||||
xml += header_data_to_xml(rows, cols, dtypes, col_bounds, col_to_format, df, dim)
|
||||
|
||||
# we already have here formatted_row_elements, so we pass here %s as a default format
|
||||
xml += array_data_to_xml(rows, cols, formatted_row_elements, format='%s')
|
||||
return xml
|
||||
|
||||
def dataset_to_xml(dataset, name, roffset, coffset, rows, cols, format):
|
||||
return dataframe_to_xml(dataset.to_pandas(), name, roffset, coffset, rows, cols, format)
|
||||
|
||||
|
||||
def array_data_to_xml(rows, cols, get_row, format):
|
||||
xml = "<arraydata rows=\"%s\" cols=\"%s\"/>\n" % (rows, cols)
|
||||
for row in range(rows):
|
||||
xml += "<row index=\"%s\"/>\n" % row
|
||||
for value in get_row(row):
|
||||
xml += var_to_xml(value, '', format=format)
|
||||
return xml
|
||||
|
||||
|
||||
def slice_to_xml(slice, rows, cols, format, type, bounds):
|
||||
return '<array slice=\"%s\" rows=\"%s\" cols=\"%s\" format=\"%s\" type=\"%s\" max=\"%s\" min=\"%s\"/>' % \
|
||||
(quote(slice), rows, cols, quote(format), type, quote(str(bounds[1])), quote(str(bounds[0])))
|
||||
|
||||
|
||||
def header_data_to_xml(rows, cols, dtypes, col_bounds, col_to_format, df, dim):
|
||||
xml = "<headerdata rows=\"%s\" cols=\"%s\">\n" % (rows, cols)
|
||||
for col in range(cols):
|
||||
col_label = quote(get_label(df.axes[1][col]) if dim > 1 else str(col))
|
||||
bounds = col_bounds[col]
|
||||
col_format = "%" + col_to_format(dtypes[col])
|
||||
xml += '<colheader index=\"%s\" label=\"%s\" type=\"%s\" format=\"%s\" max=\"%s\" min=\"%s\" />\n' % \
|
||||
(str(col), col_label, dtypes[col], col_to_format(dtypes[col]), quote(str(col_format % bounds[1])), quote(str(col_format % bounds[0])))
|
||||
for row in range(rows):
|
||||
xml += "<rowheader index=\"%s\" label = \"%s\"/>\n" % (str(row), quote(get_label(df.axes[0][row] if dim != -1 else str(row))))
|
||||
xml += "</headerdata>\n"
|
||||
return xml
|
||||
|
||||
|
||||
def is_able_to_format_number(format):
|
||||
try:
|
||||
format % math.pi
|
||||
except Exception:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
TYPE_TO_XML_CONVERTERS = {
|
||||
"ndarray": array_to_xml,
|
||||
"recarray": array_to_xml,
|
||||
"DataFrame": dataframe_to_xml,
|
||||
"Series": dataframe_to_xml,
|
||||
"GeoDataFrame": dataframe_to_xml,
|
||||
"GeoSeries": dataframe_to_xml,
|
||||
"EagerTensor": tf_to_xml,
|
||||
"ResourceVariable": tf_to_xml,
|
||||
"SparseTensor": tf_sparse_to_xml,
|
||||
"Tensor": torch_to_xml,
|
||||
"Dataset": dataset_to_xml
|
||||
}
|
||||
|
||||
|
||||
def table_like_struct_to_xml(array, name, roffset, coffset, rows, cols, format):
|
||||
_, type_name, _ = get_type(array)
|
||||
format = format if is_able_to_format_number(format) else '%'
|
||||
if type_name in TYPE_TO_XML_CONVERTERS:
|
||||
return "<xml>%s</xml>" % TYPE_TO_XML_CONVERTERS[type_name](array, name, roffset, coffset, rows, cols, format)
|
||||
else:
|
||||
raise VariableError("type %s not supported" % type_name)
|
||||
@@ -0,0 +1 @@
|
||||
# Copyright 2000-2023 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
+1
@@ -0,0 +1 @@
|
||||
# Copyright 2000-2025 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
+54
@@ -0,0 +1,54 @@
|
||||
# Copyright 2000-2025 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
import base64
|
||||
import io
|
||||
import uuid
|
||||
|
||||
IMAGE_DATA_STORAGE = {}
|
||||
DEFAULT_IMAGE_FORMAT = 'PNG'
|
||||
DEFAULT_ENCODING = 'utf-8'
|
||||
GRAYSCALE_MODE = 'L'
|
||||
RGB_MODE = 'RGB'
|
||||
RGBA_MODE = 'RGBA'
|
||||
CHUNK_SIZE = 8192
|
||||
|
||||
def load_image_chunk(offset, image_id):
|
||||
# type: (int, str) -> str
|
||||
try:
|
||||
bytes_data = IMAGE_DATA_STORAGE.get(image_id)
|
||||
if bytes_data is None:
|
||||
return "Error: No image data found."
|
||||
chunk = bytes_data[offset:offset + CHUNK_SIZE]
|
||||
next_offset = offset + CHUNK_SIZE
|
||||
if next_offset >= len(bytes_data):
|
||||
next_offset = -1
|
||||
IMAGE_DATA_STORAGE.pop(image_id, None)
|
||||
chunk_bytes = base64.b64encode(chunk)
|
||||
if not isinstance(chunk_bytes, str):
|
||||
chunk_bytes = chunk_bytes.decode(DEFAULT_ENCODING)
|
||||
return "{};{}".format(chunk_bytes, next_offset)
|
||||
except ValueError:
|
||||
return "Error: Invalid offset format."
|
||||
except Exception as e:
|
||||
return "Error: {}".format(e)
|
||||
|
||||
|
||||
def save_image_to_storage(image_data, data_type=None, format=DEFAULT_IMAGE_FORMAT, save_func=None):
|
||||
# type: (any, str, str, callable) -> str
|
||||
try:
|
||||
bytes_buffer = io.BytesIO()
|
||||
try:
|
||||
if save_func:
|
||||
save_func(bytes_buffer, format)
|
||||
else:
|
||||
image_data.save(bytes_buffer, format=format)
|
||||
bytes_buffer.seek(0)
|
||||
bytes_data = bytes_buffer.getvalue()
|
||||
image_id = str(uuid.uuid4())
|
||||
IMAGE_DATA_STORAGE[image_id] = bytes_data
|
||||
if data_type is None:
|
||||
data_type = "None"
|
||||
return "{};{};{}".format(image_id, len(bytes_data), data_type)
|
||||
finally:
|
||||
bytes_buffer.close()
|
||||
except Exception as e:
|
||||
return "Error: {}".format(e)
|
||||
+25
@@ -0,0 +1,25 @@
|
||||
# Copyright 2000-2025 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
from _pydevd_bundle.tables.images.pydevd_image_loader import save_image_to_storage, DEFAULT_IMAGE_FORMAT
|
||||
|
||||
|
||||
def create_image(figure):
|
||||
# type: (Union[matplotlib.figure.Figure | plotly.graph_objs._figure.Figure]) -> str
|
||||
try:
|
||||
try:
|
||||
import matplotlib.figure
|
||||
except ImportError:
|
||||
matplotlib = None
|
||||
|
||||
try:
|
||||
from plotly.graph_objects import Figure as PlotlyFigure
|
||||
except ImportError:
|
||||
PlotlyFigure = None
|
||||
|
||||
if matplotlib and isinstance(figure, matplotlib.figure.Figure):
|
||||
return save_image_to_storage(figure, format=DEFAULT_IMAGE_FORMAT, save_func=lambda buffer, fmt: figure.savefig(buffer, format=fmt))
|
||||
elif PlotlyFigure and isinstance(figure, PlotlyFigure):
|
||||
return save_image_to_storage(figure, format=DEFAULT_IMAGE_FORMAT, save_func=lambda buffer, fmt: buffer.write(figure.to_image(format=fmt)))
|
||||
else:
|
||||
return "Error: Unsupported figure type."
|
||||
except Exception as e:
|
||||
return "Error: {}".format(e)
|
||||
+103
@@ -0,0 +1,103 @@
|
||||
# Copyright 2000-2025 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
import numpy as np
|
||||
from _pydevd_bundle.tables.images.pydevd_image_loader import (save_image_to_storage, GRAYSCALE_MODE, RGB_MODE, RGBA_MODE)
|
||||
|
||||
try:
|
||||
import tensorflow as tf
|
||||
except ImportError:
|
||||
pass
|
||||
try:
|
||||
import torch
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
|
||||
MAX_PIXELS = 144_000_000
|
||||
|
||||
def create_image(arr):
|
||||
# type: (np.ndarray) -> str
|
||||
try:
|
||||
from PIL import Image
|
||||
|
||||
if hasattr(arr.dtype, 'name'):
|
||||
data_type = arr.dtype.name
|
||||
else:
|
||||
data_type = arr.dtype
|
||||
arr_to_convert = arr
|
||||
|
||||
try:
|
||||
import tensorflow as tf
|
||||
if isinstance(arr_to_convert, tf.SparseTensor):
|
||||
arr_to_convert = tf.sparse.to_dense(tf.sparse.reorder(arr_to_convert))
|
||||
except ImportError:
|
||||
pass
|
||||
try:
|
||||
import torch
|
||||
if isinstance(arr_to_convert, torch.Tensor):
|
||||
if arr_to_convert.requires_grad:
|
||||
arr_to_convert = arr_to_convert.detach()
|
||||
arr_to_convert = arr_to_convert.to_dense()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
arr_to_convert = arr_to_convert.numpy()
|
||||
arr_to_convert = np.where(arr_to_convert == None, 0, arr_to_convert)
|
||||
arr_to_convert = np.nan_to_num(arr_to_convert, nan=0, posinf=255, neginf=0)
|
||||
|
||||
if np.iscomplexobj(arr_to_convert) or np.issubdtype(arr_to_convert.dtype, np.timedelta64):
|
||||
raise ValueError("Only non-complex numeric array types are supported.")
|
||||
|
||||
if arr_to_convert.ndim == 1:
|
||||
arr_to_convert = np.expand_dims(arr_to_convert, axis=0)
|
||||
elif arr_to_convert.ndim == 3 and arr_to_convert.shape[2] == 1:
|
||||
arr_to_convert = arr_to_convert[:, :, 0]
|
||||
|
||||
h, w = arr_to_convert.shape[:2]
|
||||
channels = arr_to_convert.shape[2] if arr_to_convert.ndim == 3 else 1
|
||||
total_pixels = h * w * channels
|
||||
if total_pixels > MAX_PIXELS:
|
||||
scale = (MAX_PIXELS / total_pixels) ** 0.5
|
||||
new_h, new_w = max(1, int(h * scale)), max(1, int(w * scale))
|
||||
arr_to_convert = average_pooling(arr_to_convert, new_h, new_w)
|
||||
|
||||
arr_min, arr_max = arr_to_convert.min(), arr_to_convert.max()
|
||||
is_float = np.issubdtype(arr_to_convert.dtype, np.floating)
|
||||
is_bool = np.issubdtype(arr_to_convert.dtype, np.bool_)
|
||||
|
||||
if (is_float or is_bool) and 0 <= arr_min <= 1 and 0 <= arr_max <= 1: # bool and float in [0; 1]
|
||||
arr_to_convert = (arr_to_convert * 255).astype(np.uint8)
|
||||
elif arr_min != arr_max and (arr_min < 0 or arr_max > 255): # other values out of [0; 255]
|
||||
arr_to_convert = ((arr_to_convert - arr_min) * 255 / (arr_max - arr_min)).astype(np.uint8)
|
||||
elif arr_min == arr_max and (arr_min < 0 or arr_max > 255):
|
||||
arr_to_convert = (np.ones_like(arr_to_convert) * 127).astype(np.uint8)
|
||||
else: # values in [0; 255]
|
||||
arr_to_convert = arr_to_convert.astype(np.uint8)
|
||||
|
||||
if arr_to_convert.ndim == 2:
|
||||
mode = GRAYSCALE_MODE
|
||||
elif arr_to_convert.ndim == 3 and arr_to_convert.shape[2] == 4:
|
||||
mode = RGBA_MODE
|
||||
else:
|
||||
mode = RGB_MODE
|
||||
|
||||
return save_image_to_storage(Image.fromarray(arr_to_convert, mode=mode), data_type=data_type)
|
||||
|
||||
except ImportError:
|
||||
return "Error: Pillow library is not installed."
|
||||
except (TypeError, ValueError):
|
||||
return "Error: Only non-complex numeric array types are supported."
|
||||
except Exception as e:
|
||||
return "Error: {}".format(e)
|
||||
|
||||
|
||||
def average_pooling(arr, target_h, target_w):
|
||||
# type: (np.ndarray, int, int) -> np.ndarray
|
||||
h, w = arr.shape[:2]
|
||||
factor_h, factor_w = int(h / target_h), int(w / target_w)
|
||||
arr_cropped = arr[:target_h * factor_h, :target_w * factor_w]
|
||||
|
||||
if arr.ndim == 2:
|
||||
reshaped = arr_cropped.reshape(target_h, factor_h, target_w, factor_w)
|
||||
elif arr.ndim == 3:
|
||||
reshaped = arr_cropped.reshape(target_h, factor_h, target_w, factor_w, arr.shape[2])
|
||||
return reshaped.mean(axis=(1, 3))
|
||||
+79
@@ -0,0 +1,79 @@
|
||||
# Copyright 2000-2025 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
import numpy as np
|
||||
from _pydevd_bundle.tables.images.pydevd_image_loader import (save_image_to_storage, GRAYSCALE_MODE, RGB_MODE, RGBA_MODE)
|
||||
|
||||
MAX_PIXELS = 144_000_000
|
||||
MAX_DIMENSION = 15_000
|
||||
|
||||
def create_image(arr):
|
||||
# type: (np.ndarray) -> str
|
||||
try:
|
||||
from PIL import Image
|
||||
|
||||
data_type = arr.dtype.name
|
||||
arr_to_convert = arr
|
||||
|
||||
arr_to_convert = np.where(arr_to_convert == None, 0, arr_to_convert)
|
||||
arr_to_convert = np.nan_to_num(arr_to_convert, nan=0, posinf=255, neginf=0)
|
||||
|
||||
if np.iscomplexobj(arr_to_convert) or np.issubdtype(arr_to_convert.dtype, np.timedelta64):
|
||||
raise ValueError("Only non-complex numeric array types are supported.")
|
||||
|
||||
if arr_to_convert.ndim == 1:
|
||||
arr_to_convert = np.expand_dims(arr_to_convert, axis=0)
|
||||
elif arr_to_convert.ndim == 3 and arr_to_convert.shape[2] == 1:
|
||||
arr_to_convert = arr_to_convert[:, :, 0]
|
||||
|
||||
h, w = arr_to_convert.shape[:2]
|
||||
channels = arr_to_convert.shape[2] if arr_to_convert.ndim == 3 else 1
|
||||
total_pixels = h * w * channels
|
||||
if (total_pixels > MAX_PIXELS) or (h > MAX_DIMENSION) or (w > MAX_DIMENSION):
|
||||
scale_h = min(1.0, MAX_DIMENSION / float(h))
|
||||
scale_w = min(1.0, MAX_DIMENSION / float(w))
|
||||
scale_p = (MAX_PIXELS / float(total_pixels)) ** 0.5 if total_pixels > MAX_PIXELS else 1.0
|
||||
scale = min(scale_h, scale_w, scale_p)
|
||||
new_h, new_w = max(1, int(round(h * scale))), max(1, int(round(w * scale)))
|
||||
if new_h < h or new_w < w:
|
||||
arr_to_convert = average_pooling(arr_to_convert, new_h, new_w)
|
||||
|
||||
arr_min, arr_max = arr_to_convert.min(), arr_to_convert.max()
|
||||
is_float = np.issubdtype(arr_to_convert.dtype, np.floating)
|
||||
is_bool = np.issubdtype(arr_to_convert.dtype, np.bool_)
|
||||
|
||||
if (is_float or is_bool) and 0 <= arr_min <= 1 and 0 <= arr_max <= 1: # bool and float in [0; 1]
|
||||
arr_to_convert = (arr_to_convert * 255).astype(np.uint8)
|
||||
elif arr_min != arr_max and (arr_min < 0 or arr_max > 255):
|
||||
arr_to_convert = ((arr_to_convert - arr_min) * 255 / (arr_max - arr_min)).astype(np.uint8) # other values out of [0; 255]
|
||||
elif arr_min == arr_max and (arr_min < 0 or arr_max > 255):
|
||||
arr_to_convert = (np.ones_like(arr_to_convert) * 127).astype(np.uint8)
|
||||
else: # values in [0; 255]
|
||||
arr_to_convert = arr_to_convert.astype(np.uint8)
|
||||
|
||||
if arr_to_convert.ndim == 2:
|
||||
mode = GRAYSCALE_MODE
|
||||
elif arr_to_convert.ndim == 3 and arr_to_convert.shape[2] == 4:
|
||||
mode = RGBA_MODE
|
||||
else:
|
||||
mode = RGB_MODE
|
||||
|
||||
return save_image_to_storage(Image.fromarray(arr_to_convert, mode=mode), data_type=data_type)
|
||||
|
||||
except ImportError:
|
||||
return "Error: Pillow library is not installed."
|
||||
except (TypeError, ValueError):
|
||||
return "Error: Only non-complex numeric array types are supported."
|
||||
except Exception as e:
|
||||
return "Error: {}".format(e)
|
||||
|
||||
|
||||
def average_pooling(arr, target_h, target_w):
|
||||
# type: (np.ndarray, int, int) -> np.ndarray
|
||||
h, w = arr.shape[:2]
|
||||
factor_h, factor_w = int(h / target_h), int(w / target_w)
|
||||
arr_cropped = arr[:target_h * factor_h, :target_w * factor_w]
|
||||
|
||||
if arr.ndim == 2:
|
||||
reshaped = arr_cropped.reshape(target_h, factor_h, target_w, factor_w)
|
||||
elif arr.ndim == 3:
|
||||
reshaped = arr_cropped.reshape(target_h, factor_h, target_w, factor_w, arr.shape[2])
|
||||
return reshaped.mean(axis=(1, 3))
|
||||
+8
@@ -0,0 +1,8 @@
|
||||
# Copyright 2000-2025 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
import PIL
|
||||
from _pydevd_bundle.tables.images.pydevd_image_loader import save_image_to_storage, DEFAULT_IMAGE_FORMAT
|
||||
|
||||
def create_image(pillow_image):
|
||||
# type: (PIL.Image.Image) -> str
|
||||
image_format = pillow_image.format if pillow_image.format else DEFAULT_IMAGE_FORMAT
|
||||
return save_image_to_storage(pillow_image, format=image_format)
|
||||
+157
@@ -0,0 +1,157 @@
|
||||
# Copyright 2000-2023 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
import pandas as pd
|
||||
|
||||
TABLE_TYPE_NEXT_VALUE_SEPARATOR = '__pydev_table_column_type_val__'
|
||||
MAX_COLWIDTH_PYTHON_2 = 100000
|
||||
BATCH_SIZE = 10000
|
||||
|
||||
CSV_FORMAT_SEPARATOR = '~'
|
||||
|
||||
|
||||
def get_type(table):
|
||||
# type: (str) -> str
|
||||
return str(type(table))
|
||||
|
||||
|
||||
# noinspection PyUnresolvedReferences
|
||||
def get_shape(table):
|
||||
# type: (datasets.arrow_dataset.Dataset) -> str
|
||||
return str(table.shape[0])
|
||||
|
||||
|
||||
# noinspection PyUnresolvedReferences
|
||||
def get_head(table):
|
||||
# type: (datasets.arrow_dataset.Dataset) -> str
|
||||
return repr(__convert_to_df(table.select([0])).head(1).to_html(notebook=True))
|
||||
|
||||
|
||||
# noinspection PyUnresolvedReferences
|
||||
def get_column_types(table):
|
||||
# type: (datasets.arrow_dataset.Dataset) -> str
|
||||
table = __convert_to_df(table.select([0]))
|
||||
return str(table.index.dtype) + TABLE_TYPE_NEXT_VALUE_SEPARATOR + \
|
||||
TABLE_TYPE_NEXT_VALUE_SEPARATOR.join([str(t) for t in table.dtypes])
|
||||
|
||||
|
||||
# used by pydevd
|
||||
# noinspection PyUnresolvedReferences
|
||||
def get_data(table, use_csv_serialization, start_index=None, end_index=None, format=None):
|
||||
# type: (datasets.arrow_dataset.Dataset, int, int) -> str
|
||||
|
||||
def convert_data_to_csv(data, format):
|
||||
return repr(__convert_to_df(data).to_csv(na_rep = "NaN", float_format=format, sep=CSV_FORMAT_SEPARATOR))
|
||||
|
||||
def convert_data_to_html(data, format):
|
||||
return repr(__convert_to_df(data).to_html(notebook=True))
|
||||
|
||||
if use_csv_serialization:
|
||||
computed_data = __compute_sliced_data(table, convert_data_to_csv, start_index, end_index, format)
|
||||
else:
|
||||
computed_data = __compute_sliced_data(table, convert_data_to_html, start_index, end_index, format)
|
||||
return computed_data
|
||||
|
||||
|
||||
# used by DSTableCommands
|
||||
# noinspection PyUnresolvedReferences
|
||||
def display_data_html(table, start_index, end_index):
|
||||
# type: (datasets.arrow_dataset.Dataset, int, int) -> None
|
||||
def ipython_display(data, format):
|
||||
from IPython.display import display, HTML
|
||||
display(HTML(__convert_to_df(data).to_html(notebook=True)))
|
||||
__compute_sliced_data(table, ipython_display, start_index, end_index)
|
||||
|
||||
|
||||
# used by DSTableCommands
|
||||
# noinspection PyUnresolvedReferences
|
||||
def display_data_csv(table, start_index, end_index):
|
||||
# type: (datasets.arrow_dataset.Dataset, int, int) -> None
|
||||
def ipython_display(data, format):
|
||||
try:
|
||||
data = data.to_csv(na_rep = "NaN", sep=CSV_FORMAT_SEPARATOR, float_format=format)
|
||||
except AttributeError:
|
||||
pass
|
||||
print(repr(__convert_to_df(data)))
|
||||
__compute_sliced_data(table, ipython_display, start_index, end_index)
|
||||
|
||||
|
||||
def __get_data_slice(table, start, end):
|
||||
# type: (datasets.arrow_dataset.Dataset, int, int) -> pd.DataFrame
|
||||
return __convert_to_df(table).iloc[start:end]
|
||||
|
||||
|
||||
def __compute_sliced_data(table, fun, start_index=None, end_index=None, format=None):
|
||||
# type: (datasets.arrow_dataset.Dataset, function, int, int) -> str
|
||||
max_cols, max_colwidth, max_rows = __get_tables_display_options()
|
||||
|
||||
_jb_max_cols = pd.get_option('display.max_columns')
|
||||
_jb_max_colwidth = pd.get_option('display.max_colwidth')
|
||||
_jb_max_rows = pd.get_option('display.max_rows')
|
||||
if format is not None:
|
||||
_jb_float_options = pd.get_option('display.float_format')
|
||||
|
||||
pd.set_option('display.max_columns', max_cols)
|
||||
pd.set_option('display.max_rows', max_rows)
|
||||
pd.set_option('display.max_colwidth', max_colwidth)
|
||||
|
||||
format_function = __define_format_function(format)
|
||||
if format_function is not None:
|
||||
pd.set_option('display.float_format', format_function)
|
||||
|
||||
if start_index is not None and end_index is not None:
|
||||
table = __get_data_slice(table, start_index, end_index)
|
||||
|
||||
data = fun(table, pd.get_option('display.float_format'))
|
||||
|
||||
pd.set_option('display.max_columns', _jb_max_cols)
|
||||
pd.set_option('display.max_colwidth', _jb_max_colwidth)
|
||||
pd.set_option('display.max_rows', _jb_max_rows)
|
||||
if format is not None:
|
||||
pd.set_option('display.float_format', _jb_float_options)
|
||||
|
||||
return data
|
||||
|
||||
|
||||
def __define_format_function(format):
|
||||
# type: (Union[None, str]) -> Union[Callable, None]
|
||||
if format is None or format == 'null':
|
||||
return None
|
||||
|
||||
if type(format) == str and format.startswith("%"):
|
||||
return lambda x: format % x
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def __get_tables_display_options():
|
||||
# type: () -> Tuple[None, Union[int, None], None]
|
||||
try:
|
||||
# In pandas versions earlier than 1.0, max_colwidth must be set as an integer
|
||||
if int(pd.__version__.split('.')[0]) < 1:
|
||||
return None, MAX_COLWIDTH_PYTHON_2, None
|
||||
except ImportError:
|
||||
pass
|
||||
return None, None, None
|
||||
|
||||
|
||||
# noinspection PyUnresolvedReferences
|
||||
def __convert_to_df(table):
|
||||
# type: (datasets.arrow_dataset.Dataset) -> pd.DataFrame
|
||||
try:
|
||||
import datasets
|
||||
if type(table) is datasets.arrow_dataset.Dataset:
|
||||
return __dataset_to_df(table)
|
||||
except ImportError as e:
|
||||
pass
|
||||
return table
|
||||
|
||||
|
||||
def __dataset_to_df(dataset):
|
||||
# type: (datasets.arrow_dataset.Dataset) -> pd.DataFrame
|
||||
try:
|
||||
dataset_as_df = list(dataset.to_pandas(batched=True, batch_size=min(len(dataset), BATCH_SIZE)))
|
||||
if len(dataset_as_df) > 1:
|
||||
return pd.concat(dataset_as_df, ignore_index=True)
|
||||
else:
|
||||
return dataset_as_df[0]
|
||||
except ImportError as e:
|
||||
pass
|
||||
@@ -0,0 +1,405 @@
|
||||
# Copyright 2000-2023 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
import io
|
||||
import numpy as np
|
||||
|
||||
TABLE_TYPE_NEXT_VALUE_SEPARATOR = '__pydev_table_column_type_val__'
|
||||
MAX_COLWIDTH = 100000
|
||||
|
||||
ONE_DIM, TWO_DIM, WITH_TYPES = range(3)
|
||||
NP_ROWS_TYPE = "int64"
|
||||
|
||||
CSV_FORMAT_SEPARATOR = '~'
|
||||
|
||||
is_pd = False
|
||||
try:
|
||||
import pandas as pd
|
||||
is_pd = True
|
||||
except:
|
||||
pass
|
||||
|
||||
|
||||
def get_type(table):
|
||||
# type: (np.ndarray) -> str
|
||||
return str(type(table))
|
||||
|
||||
|
||||
def get_shape(table):
|
||||
# type: (np.ndarray) -> str
|
||||
if table.dtype.names is not None:
|
||||
return str((table.shape[0], len(table.dtype.names)))
|
||||
if table.ndim == 1:
|
||||
return str((table.shape[0], 1))
|
||||
elif table.ndim == 0:
|
||||
return str((0, 0))
|
||||
else:
|
||||
return str((table.shape[0], table.shape[1]))
|
||||
|
||||
|
||||
def get_head(table):
|
||||
# type: (np.ndarray) -> str
|
||||
column_names = table.dtype.names
|
||||
if column_names:
|
||||
return TABLE_TYPE_NEXT_VALUE_SEPARATOR.join([str(column_names[i]) for i in range(len(column_names))])
|
||||
return "None"
|
||||
|
||||
|
||||
def get_column_types(table):
|
||||
# type: (np.ndarray) -> str
|
||||
if table.ndim == 0:
|
||||
return ""
|
||||
table = __create_table(table[:1])
|
||||
try:
|
||||
cols_types = [str(t) for t in table.dtypes] if is_pd else table.get_cols_types()
|
||||
except AttributeError:
|
||||
cols_types = table.get_cols_types()
|
||||
|
||||
return NP_ROWS_TYPE + TABLE_TYPE_NEXT_VALUE_SEPARATOR + \
|
||||
TABLE_TYPE_NEXT_VALUE_SEPARATOR.join(cols_types)
|
||||
|
||||
|
||||
def get_data(table, use_csv_serialization, start_index=None, end_index=None, format=None):
|
||||
# type: (Union[np.ndarray, dict], int, int) -> str
|
||||
def convert_data_to_html(data, format):
|
||||
return repr(__create_table(data, start_index, end_index, format).to_html(notebook=True))
|
||||
|
||||
def convert_data_to_csv(data, format):
|
||||
return repr(__create_table(data, start_index, end_index, format).to_csv(na_rep ="None", float_format=format, sep=CSV_FORMAT_SEPARATOR))
|
||||
|
||||
if use_csv_serialization:
|
||||
computed_data = __compute_data(table, convert_data_to_csv, format)
|
||||
else:
|
||||
computed_data = __compute_data(table, convert_data_to_html, format)
|
||||
return computed_data
|
||||
|
||||
|
||||
def display_data_html(table, start_index=None, end_index=None):
|
||||
# type: (np.ndarray, int, int) -> None
|
||||
def ipython_display(data, format):
|
||||
from IPython.display import display, HTML
|
||||
display(HTML(__create_table(data, start_index, end_index).to_html(notebook=True)))
|
||||
|
||||
__compute_data(table, ipython_display)
|
||||
|
||||
|
||||
def display_data_csv(table, start_index=None, end_index=None):
|
||||
# type: (np.ndarray, int, int) -> None
|
||||
def ipython_display(data, format):
|
||||
print(repr(__create_table(data, start_index, end_index).to_csv(na_rep ="None", sep=CSV_FORMAT_SEPARATOR, float_format=format)))
|
||||
|
||||
__compute_data(table, ipython_display)
|
||||
|
||||
|
||||
def remove_nones_values(array_part, na_rep):
|
||||
if np.issubdtype(array_part.dtype, np.number):
|
||||
array_part_without_nones = np.where(array_part == None, np.nan, array_part)
|
||||
else:
|
||||
array_part_without_nones = np.where(array_part == None, na_rep, array_part)
|
||||
return array_part_without_nones
|
||||
|
||||
|
||||
class _NpTable:
|
||||
def __init__(self, np_array, format=None):
|
||||
self.array = np_array
|
||||
self.type = self.get_array_type()
|
||||
self.indexes = None
|
||||
self.format = format
|
||||
|
||||
def get_array_type(self):
|
||||
col_type = self.array.dtype
|
||||
|
||||
if len(col_type) != 0:
|
||||
return WITH_TYPES
|
||||
|
||||
if self.array.ndim > 1:
|
||||
return TWO_DIM
|
||||
|
||||
return ONE_DIM
|
||||
|
||||
def get_cols_types(self):
|
||||
col_type = self.array.dtype
|
||||
|
||||
if self.type == ONE_DIM:
|
||||
# [1, 2, 3] -> [int]
|
||||
return [str(col_type)]
|
||||
|
||||
if self.type == WITH_TYPES:
|
||||
# ([(10, 3.14), (20, 2.71)], dtype=[("ci", "i4"), ("cf", "f4")]) -> [int, float]
|
||||
return [str(col_type[i]) for i in range(len(col_type))] # is not iterable
|
||||
|
||||
# [[1, 2], [3, 4]] -> [int, int]
|
||||
return [str(col_type) for _ in range(len(self.array[0]))]
|
||||
|
||||
def head(self, num_rows):
|
||||
if self.array.shape[0] < 6:
|
||||
return self
|
||||
|
||||
return _NpTable(self.array[:5]).sort()
|
||||
|
||||
def to_html(self, notebook):
|
||||
html = ['<table class="dataframe">\n']
|
||||
|
||||
# columns names
|
||||
html.append('<thead>\n'
|
||||
'<tr style="text-align: right;">\n'
|
||||
'<th></th>\n')
|
||||
html += self.__collect_cols_names()
|
||||
html.append('</tr>\n'
|
||||
'</thead>\n')
|
||||
|
||||
# tbody
|
||||
html += self.__collect_values(None)
|
||||
|
||||
html.append('</table>\n')
|
||||
|
||||
return "".join(html)
|
||||
|
||||
def __collect_cols_names(self):
|
||||
if self.type == ONE_DIM:
|
||||
return ['<th>0</th>\n']
|
||||
|
||||
if self.type == WITH_TYPES:
|
||||
columns_names = self.array.dtype.names
|
||||
return ['<th>{}</th>\n'.format(str(columns_names[i])) for i in range(len(columns_names))]
|
||||
|
||||
return ['<th>{}</th>\n'.format(i) for i in range(len(self.array[0]))]
|
||||
|
||||
def __collect_values(self, max_cols):
|
||||
html = ['<tbody>\n']
|
||||
rows = self.array.shape[0]
|
||||
for row_num in range(rows):
|
||||
html.append('<tr>\n')
|
||||
html.append('<th>{}</th>\n'.format(int(self.indexes[row_num])))
|
||||
if self.type == ONE_DIM:
|
||||
if self.format is not None and self.array[row_num] is not None and self.array[row_num] == self.array[row_num]:
|
||||
try:
|
||||
value = self.format % self.array[row_num]
|
||||
except Exception as _:
|
||||
value = self.array[row_num]
|
||||
else:
|
||||
value = self.array[row_num]
|
||||
html.append('<td>{}</td>\n'.format(value))
|
||||
else:
|
||||
cols = len(self.array[0])
|
||||
max_cols = cols if max_cols is None else min(max_cols, cols)
|
||||
for col_num in range(max_cols):
|
||||
if self.format is not None and self.array[row_num][col_num] is not None and self.array[row_num][col_num] == self.array[row_num][col_num]:
|
||||
try:
|
||||
value = self.format % self.array[row_num][col_num]
|
||||
except Exception as _:
|
||||
value = self.array[row_num][col_num]
|
||||
else:
|
||||
value = self.array[row_num][col_num]
|
||||
html.append('<td>{}</td>\n'.format(value))
|
||||
html.append('</tr>\n')
|
||||
html.append('</tbody>\n')
|
||||
return html
|
||||
|
||||
def to_csv(self, na_rep="None", float_format=None, sep=CSV_FORMAT_SEPARATOR):
|
||||
csv_stream = io.StringIO()
|
||||
if self.array.dtype.names is not None:
|
||||
np_array_without_nones = []
|
||||
for field in self.array.dtype.names:
|
||||
np_array_without_nones.append(remove_nones_values(self.array[str(field)], na_rep))
|
||||
np_array_without_nones = np.column_stack(np_array_without_nones)
|
||||
else:
|
||||
np_array_without_nones = remove_nones_values(self.array, na_rep)
|
||||
if float_format is None or float_format == 'null':
|
||||
float_format = "%s"
|
||||
|
||||
np.savetxt(csv_stream, np_array_without_nones, delimiter=sep, fmt=float_format)
|
||||
csv_string = csv_stream.getvalue()
|
||||
csv_rows_with_index = self.__insert_index_at_rows_begging_csv(csv_string)
|
||||
|
||||
col_names = self.__collect_col_names_csv()
|
||||
return col_names + "\n" + csv_rows_with_index
|
||||
|
||||
def __insert_index_at_rows_begging_csv(self, csv_string):
|
||||
# type: (str) -> str
|
||||
csv_rows = csv_string.split('\n')
|
||||
csv_rows_with_index = []
|
||||
for row_index in range(self.array.shape[0]):
|
||||
csv_rows_with_index.append(str(row_index) + CSV_FORMAT_SEPARATOR + csv_rows[row_index])
|
||||
return "\n".join(csv_rows_with_index)
|
||||
|
||||
def __collect_col_names_csv(self):
|
||||
if self.type == ONE_DIM:
|
||||
return '{}0'.format(CSV_FORMAT_SEPARATOR)
|
||||
|
||||
if self.type == WITH_TYPES:
|
||||
return CSV_FORMAT_SEPARATOR + CSV_FORMAT_SEPARATOR.join(['{}'.format(name) for name in self.array.dtype.names])
|
||||
|
||||
# TWO_DIM
|
||||
return CSV_FORMAT_SEPARATOR + CSV_FORMAT_SEPARATOR.join(['{}'.format(i) for i in range(self.array.shape[1])])
|
||||
|
||||
|
||||
def slice(self, start_index=None, end_index=None):
|
||||
if end_index is not None and start_index is not None:
|
||||
self.array = self.array[start_index:end_index]
|
||||
self.indexes = self.indexes[start_index:end_index]
|
||||
|
||||
return self
|
||||
|
||||
def sort(self, sort_keys=None):
|
||||
self.indexes = np.arange(self.array.shape[0])
|
||||
if sort_keys is None:
|
||||
return self
|
||||
|
||||
cols, orders = sort_keys
|
||||
if 0 in cols:
|
||||
return self.__sort_by_index(True in orders)
|
||||
|
||||
if self.type == ONE_DIM:
|
||||
extended = np.column_stack((self.indexes, self.array))
|
||||
sort_extended = extended[:, 1].argsort()
|
||||
if False in orders:
|
||||
sort_extended = sort_extended[::-1]
|
||||
result = extended[sort_extended]
|
||||
self.array = result[:, 1]
|
||||
self.indexes = result[:, 0]
|
||||
return self
|
||||
|
||||
if self.type == WITH_TYPES:
|
||||
new_dt = np.dtype([('_pydevd_i', 'i8')] + self.array.dtype.descr)
|
||||
extended = np.zeros(self.array.shape, dtype=new_dt)
|
||||
extended['_pydevd_i'] = list(range(self.array.shape[0]))
|
||||
for col in self.array.dtype.names:
|
||||
extended[col] = self.array[col]
|
||||
|
||||
column_names = self.array.dtype.names
|
||||
for i in range(len(cols) - 1, -1, -1):
|
||||
name = column_names[cols[i] - 1]
|
||||
sort = extended[name].argsort(kind='stable')
|
||||
extended = extended[sort if orders[i] else sort[::-1]]
|
||||
self.indexes = extended['_pydevd_i']
|
||||
for col in self.array.dtype.names:
|
||||
self.array[col] = extended[col]
|
||||
return self
|
||||
|
||||
extended = np.insert(self.array, 0, self.indexes, axis=1)
|
||||
for i in range(len(cols) - 1, -1, -1):
|
||||
sort = extended[:, cols[i]].argsort(kind='stable')
|
||||
extended = extended[sort if orders[i] else sort[::-1]]
|
||||
self.indexes = extended[:, 0]
|
||||
self.array = extended[:, 1:]
|
||||
return self
|
||||
|
||||
def __sort_by_index(self, order):
|
||||
if order:
|
||||
return self
|
||||
self.array = self.array[::-1]
|
||||
self.indexes = self.indexes[::-1]
|
||||
return self
|
||||
|
||||
|
||||
def __sort_df(dataframe, sort_keys):
|
||||
if sort_keys is None:
|
||||
return dataframe
|
||||
|
||||
cols, orders = sort_keys
|
||||
if 0 in cols:
|
||||
if len(cols) == 1:
|
||||
return dataframe.sort_index(ascending=orders[0])
|
||||
return dataframe.sort_index(level=cols, ascending=orders)
|
||||
sort_by = list(map(lambda c: dataframe.columns[c - 1], cols))
|
||||
return dataframe.sort_values(by=sort_by, ascending=orders)
|
||||
|
||||
|
||||
def __create_table(command, start_index=None, end_index=None, format=None):
|
||||
sort_keys = None
|
||||
|
||||
if type(command) is dict:
|
||||
np_array = command['data']
|
||||
sort_keys = command['sort_keys']
|
||||
else:
|
||||
np_array = command
|
||||
|
||||
if is_pd:
|
||||
sorted_df = __sort_df(pd.DataFrame(np_array), sort_keys)
|
||||
if start_index is not None and end_index is not None:
|
||||
sorted_df_slice = sorted_df.iloc[start_index:end_index]
|
||||
# to apply "format" we should not have None inside DFs
|
||||
try:
|
||||
import warnings
|
||||
with warnings.catch_warnings():
|
||||
warnings.simplefilter("ignore")
|
||||
sorted_df_slice = sorted_df_slice.fillna("None")
|
||||
except Exception as _:
|
||||
pass
|
||||
return sorted_df_slice
|
||||
return sorted_df
|
||||
|
||||
return _NpTable(np_array, format=format).sort(sort_keys).slice(start_index,
|
||||
end_index)
|
||||
|
||||
|
||||
def __compute_data(arr, fun, format=None):
|
||||
is_sort_command = type(arr) is dict
|
||||
data = arr['data'] if is_sort_command else arr
|
||||
|
||||
jb_max_cols, jb_max_colwidth, jb_max_rows, jb_float_options = None, None, None, None
|
||||
if is_pd:
|
||||
jb_max_cols, jb_max_colwidth, jb_max_rows, jb_float_options = __set_pd_options(format)
|
||||
|
||||
if is_sort_command:
|
||||
arr['data'] = data
|
||||
data = arr
|
||||
|
||||
format = pd.get_option('display.float_format') if is_pd else format
|
||||
|
||||
data = fun(data, format)
|
||||
|
||||
if is_pd:
|
||||
__reset_pd_options(jb_max_cols, jb_max_colwidth, jb_max_rows, jb_float_options)
|
||||
|
||||
return data
|
||||
|
||||
|
||||
def __get_tables_display_options():
|
||||
# type: () -> Tuple[None, Union[int, None], None]
|
||||
try:
|
||||
import pandas as pd
|
||||
# In pandas versions earlier than 1.0, max_colwidth must be set as an integer
|
||||
if int(pd.__version__.split('.')[0]) < 1:
|
||||
return None, MAX_COLWIDTH, None
|
||||
except Exception:
|
||||
pass
|
||||
return None, None, None
|
||||
|
||||
|
||||
def __set_pd_options(format):
|
||||
max_cols, max_colwidth, max_rows = __get_tables_display_options()
|
||||
_jb_float_options = None
|
||||
|
||||
_jb_max_cols = pd.get_option('display.max_columns')
|
||||
_jb_max_colwidth = pd.get_option('display.max_colwidth')
|
||||
_jb_max_rows = pd.get_option('display.max_rows')
|
||||
if format is not None:
|
||||
_jb_float_options = pd.get_option('display.float_format')
|
||||
|
||||
pd.set_option('display.max_columns', max_cols)
|
||||
pd.set_option('display.max_rows', max_rows)
|
||||
pd.set_option('display.max_colwidth', max_colwidth)
|
||||
format_function = __define_format_function(format)
|
||||
if format_function is not None:
|
||||
pd.set_option('display.float_format', format_function)
|
||||
|
||||
return _jb_max_cols, _jb_max_colwidth, _jb_max_rows, _jb_float_options
|
||||
|
||||
|
||||
def __reset_pd_options(max_cols, max_colwidth, max_rows, float_format):
|
||||
pd.set_option('display.max_columns', max_cols)
|
||||
pd.set_option('display.max_colwidth', max_colwidth)
|
||||
pd.set_option('display.max_rows', max_rows)
|
||||
if float_format is not None:
|
||||
pd.set_option('display.float_format', float_format)
|
||||
|
||||
|
||||
def __define_format_function(format):
|
||||
# type: (Union[None, str]) -> Union[Callable, None]
|
||||
if format is None or format == 'null':
|
||||
return None
|
||||
|
||||
if type(format) == str and format.startswith("%"):
|
||||
return lambda x: format % x
|
||||
else:
|
||||
return None
|
||||
+401
@@ -0,0 +1,401 @@
|
||||
# Copyright 2000-2024 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
import io
|
||||
import numpy as np
|
||||
|
||||
try:
|
||||
import tensorflow as tf
|
||||
except ImportError:
|
||||
pass
|
||||
try:
|
||||
import torch
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
TABLE_TYPE_NEXT_VALUE_SEPARATOR = '__pydev_table_column_type_val__'
|
||||
MAX_COLWIDTH = 100000
|
||||
|
||||
ONE_DIM, TWO_DIM = range(2)
|
||||
NP_ROWS_TYPE = "int64"
|
||||
|
||||
CSV_FORMAT_SEPARATOR = '~'
|
||||
|
||||
is_pd_can_be_imported = False
|
||||
try:
|
||||
import pandas as pd
|
||||
is_pd_can_be_imported = True
|
||||
except:
|
||||
pass
|
||||
|
||||
|
||||
def get_type(table):
|
||||
# type: (np.ndarray) -> str
|
||||
return str(type(table))
|
||||
|
||||
|
||||
def get_shape(table):
|
||||
# type: (np.ndarray) -> str
|
||||
shape = None
|
||||
try:
|
||||
import tensorflow as tf
|
||||
if isinstance(table, tf.SparseTensor):
|
||||
shape = table.dense_shape.numpy()
|
||||
else:
|
||||
shape = table.shape
|
||||
except Exception:
|
||||
shape = table.shape
|
||||
|
||||
if len(shape) == 1:
|
||||
return str((int(shape[0]), 1))
|
||||
else:
|
||||
return str((int(shape[0]), int(shape[1])))
|
||||
|
||||
|
||||
def get_head(table):
|
||||
# type: (np.ndarray) -> str
|
||||
return "None"
|
||||
|
||||
|
||||
def get_column_types(table):
|
||||
# type: (np.ndarray) -> str
|
||||
is_pandas = __is_pandas_can_be_used_for_array(table)
|
||||
table = __create_table(table)
|
||||
cols_types = [str(t) for t in table.dtypes] if is_pandas else table.get_cols_types()
|
||||
|
||||
return NP_ROWS_TYPE + TABLE_TYPE_NEXT_VALUE_SEPARATOR + \
|
||||
TABLE_TYPE_NEXT_VALUE_SEPARATOR.join(cols_types)
|
||||
|
||||
|
||||
def get_data(table, use_csv_serialization, start_index=None, end_index=None, format=None):
|
||||
# type: (Union[np.ndarray, dict], bool, Union[int, None], Union[int, None], Union[str, None]) -> str
|
||||
def convert_data_to_html(data, format):
|
||||
return repr(__create_table(data, start_index, end_index, format).to_html(notebook=True))
|
||||
|
||||
def convert_data_to_csv(data, format):
|
||||
return repr(__create_table(data, start_index, end_index, format).to_csv(na_rep ="None", float_format=format, sep=CSV_FORMAT_SEPARATOR))
|
||||
|
||||
if use_csv_serialization:
|
||||
computed_data = __compute_data(table, convert_data_to_csv, format)
|
||||
else:
|
||||
computed_data = __compute_data(table, convert_data_to_html, format)
|
||||
return computed_data
|
||||
|
||||
|
||||
def display_data_html(table, start_index=None, end_index=None):
|
||||
# type: (np.ndarray, int, int) -> None
|
||||
def ipython_display(data, format):
|
||||
from IPython.display import display, HTML
|
||||
display(HTML(__create_table(data, start_index, end_index).to_html(notebook=True)))
|
||||
|
||||
__compute_data(table, ipython_display)
|
||||
|
||||
|
||||
def display_data_csv(table, start_index=None, end_index=None):
|
||||
# type: (np.ndarray, int, int) -> None
|
||||
def ipython_display(data, format):
|
||||
print(repr(__create_table(data, start_index, end_index).to_csv(na_rep ="None", sep=CSV_FORMAT_SEPARATOR, float_format=format)))
|
||||
|
||||
__compute_data(table, ipython_display)
|
||||
|
||||
|
||||
class _NpTable:
|
||||
def __init__(self, np_array, format=None):
|
||||
self.array = np_array
|
||||
self.type = self.get_array_type()
|
||||
self.indexes = None
|
||||
self.format = format
|
||||
|
||||
def get_array_type(self):
|
||||
if len(self.array.shape) > 1:
|
||||
return TWO_DIM
|
||||
|
||||
return ONE_DIM
|
||||
|
||||
def get_cols_types(self):
|
||||
dtype = self.array.dtype
|
||||
if "torch" in str(dtype):
|
||||
col_type = dtype
|
||||
else:
|
||||
col_type = dtype.name
|
||||
|
||||
if self.type == ONE_DIM:
|
||||
# [1, 2, 3] -> [int]
|
||||
return [str(col_type)]
|
||||
|
||||
# [[1, 2], [3, 4]] -> [int, int]
|
||||
return [str(col_type) for _ in range(len(self.array[0]))]
|
||||
|
||||
def head(self, num_rows):
|
||||
if self.array.shape[0] < 6:
|
||||
return self
|
||||
|
||||
return _NpTable(self.array[:num_rows]).sort()
|
||||
|
||||
def to_html(self, notebook):
|
||||
html = ['<table class="dataframe">\n']
|
||||
|
||||
# columns names
|
||||
html.append('<thead>\n'
|
||||
'<tr style="text-align: right;">\n'
|
||||
'<th></th>\n')
|
||||
html += self.__collect_cols_names_html()
|
||||
html.append('</tr>\n'
|
||||
'</thead>\n')
|
||||
|
||||
# tbody
|
||||
html += self.__collect_values_html(None)
|
||||
|
||||
html.append('</table>\n')
|
||||
|
||||
return "".join(html)
|
||||
|
||||
def __collect_cols_names_html(self):
|
||||
if self.type == ONE_DIM:
|
||||
return ['<th>0</th>\n']
|
||||
|
||||
return ['<th>{}</th>\n'.format(i) for i in range(len(self.array[0]))]
|
||||
|
||||
def __collect_values_html(self, max_cols):
|
||||
html = ['<tbody>\n']
|
||||
rows = self.array.shape[0]
|
||||
for row_num in range(rows):
|
||||
html.append('<tr>\n')
|
||||
html.append('<th>{}</th>\n'.format(int(self.indexes[row_num])))
|
||||
if self.type == ONE_DIM:
|
||||
# None usually is not supported in tensors, but to be totally sure
|
||||
if self.format is not None and self.array[row_num] is not None and self.array[row_num] == self.array[row_num]:
|
||||
try:
|
||||
value = self.format % self.array[row_num]
|
||||
except Exception as _:
|
||||
value = self.array[row_num]
|
||||
else:
|
||||
value = self.array[row_num]
|
||||
html.append('<td>{}</td>\n'.format(value))
|
||||
else:
|
||||
cols = len(self.array[0])
|
||||
max_cols = cols if max_cols is None else min(max_cols, cols)
|
||||
for col_num in range(max_cols):
|
||||
if self.format is not None and self.array[row_num][col_num] is not None and self.array[row_num][col_num] == self.array[row_num][col_num]:
|
||||
try:
|
||||
value = self.format % self.array[row_num][col_num]
|
||||
except Exception as _:
|
||||
value = self.array[row_num][col_num]
|
||||
else:
|
||||
value = self.array[row_num][col_num]
|
||||
html.append('<td>{}</td>\n'.format(value))
|
||||
html.append('</tr>\n')
|
||||
html.append('</tbody>\n')
|
||||
return html
|
||||
|
||||
# TODO: won't work for not-CPU-stored arrays
|
||||
def to_csv(self, na_rep = "None", float_format=None, sep=CSV_FORMAT_SEPARATOR):
|
||||
csv_stream = io.StringIO()
|
||||
if float_format is None or float_format == 'null':
|
||||
float_format = "%s"
|
||||
|
||||
np.savetxt(csv_stream, self.array, delimiter=CSV_FORMAT_SEPARATOR, fmt=float_format)
|
||||
csv_string = csv_stream.getvalue()
|
||||
csv_rows_with_index = self.__insert_index_at_rows_begging_csv(csv_string)
|
||||
|
||||
col_names = self.__collect_col_names_csv()
|
||||
return col_names + "\n" + csv_rows_with_index
|
||||
|
||||
def __insert_index_at_rows_begging_csv(self, csv_string):
|
||||
# type: (str) -> str
|
||||
csv_rows = csv_string.split('\n')
|
||||
csv_rows_with_index = []
|
||||
for row_index in range(self.array.shape[0]):
|
||||
csv_rows_with_index.append(str(row_index) + CSV_FORMAT_SEPARATOR + csv_rows[row_index])
|
||||
return "\n".join(csv_rows_with_index)
|
||||
|
||||
def __collect_col_names_csv(self):
|
||||
if self.type == ONE_DIM:
|
||||
return '{}0'.format(CSV_FORMAT_SEPARATOR)
|
||||
|
||||
# TWO_DIM
|
||||
return CSV_FORMAT_SEPARATOR + CSV_FORMAT_SEPARATOR.join(['{}'.format(i) for i in range(self.array.shape[1])])
|
||||
|
||||
def slice(self, start_index=None, end_index=None):
|
||||
if end_index is not None and start_index is not None:
|
||||
self.array = self.array[start_index:end_index]
|
||||
self.indexes = self.indexes[start_index:end_index]
|
||||
|
||||
return self
|
||||
|
||||
def sort(self, sort_keys=None):
|
||||
self.indexes = np.arange(self.array.shape[0])
|
||||
if sort_keys is None:
|
||||
return self
|
||||
|
||||
cols, orders = sort_keys
|
||||
if 0 in cols:
|
||||
return self.__sort_by_index(True in orders)
|
||||
|
||||
if self.type == ONE_DIM:
|
||||
extended = np.column_stack((self.indexes, self.array))
|
||||
sort_extended = extended[:, 1].argsort()
|
||||
if False in orders:
|
||||
sort_extended = sort_extended[::-1]
|
||||
result = extended[sort_extended]
|
||||
self.array = result[:, 1]
|
||||
self.indexes = result[:, 0]
|
||||
return self
|
||||
|
||||
extended = np.insert(self.array, 0, self.indexes, axis=1)
|
||||
for i in range(len(cols) - 1, -1, -1):
|
||||
sort = extended[:, cols[i]].argsort(kind='stable')
|
||||
extended = extended[sort if orders[i] else sort[::-1]]
|
||||
self.indexes = extended[:, 0]
|
||||
self.array = extended[:, 1:]
|
||||
return self
|
||||
|
||||
def __sort_by_index(self, order):
|
||||
if order:
|
||||
return self
|
||||
self.array = self.array[::-1]
|
||||
self.indexes = self.indexes[::-1]
|
||||
return self
|
||||
|
||||
|
||||
def __sort_df(dataframe, sort_keys):
|
||||
if sort_keys is None:
|
||||
return dataframe
|
||||
|
||||
cols, orders = sort_keys
|
||||
if 0 in cols:
|
||||
if len(cols) == 1:
|
||||
return dataframe.sort_index(ascending=orders[0])
|
||||
return dataframe.sort_index(level=cols, ascending=orders)
|
||||
sort_by = list(map(lambda c: dataframe.columns[c - 1], cols))
|
||||
return dataframe.sort_values(by=sort_by, ascending=orders)
|
||||
|
||||
|
||||
def __create_table(command, start_index=None, end_index=None, format=None):
|
||||
sort_keys = None
|
||||
|
||||
if type(command) is dict:
|
||||
np_array = command['data']
|
||||
sort_keys = command['sort_keys']
|
||||
else:
|
||||
np_array = command
|
||||
|
||||
try:
|
||||
import tensorflow as tf
|
||||
if isinstance(np_array, tf.SparseTensor):
|
||||
np_array = tf.sparse.to_dense(tf.sparse.reorder(np_array))
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
import torch
|
||||
if isinstance(np_array, torch.Tensor):
|
||||
if np_array.requires_grad:
|
||||
np_array = np_array.detach()
|
||||
np_array = np_array.to_dense()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
is_pandas = __is_pandas_can_be_used_for_array(np_array)
|
||||
if is_pandas:
|
||||
sorting_arr = __sort_df(pd.DataFrame(np_array), sort_keys)
|
||||
if start_index is not None and end_index is not None:
|
||||
return sorting_arr.iloc[start_index:end_index]
|
||||
return sorting_arr
|
||||
|
||||
return _NpTable(np_array, format=format).sort(sort_keys).slice(start_index, end_index)
|
||||
|
||||
|
||||
def __compute_data(arr, fun, format=None):
|
||||
is_sort_command = type(arr) is dict
|
||||
data = arr['data'] if is_sort_command else arr
|
||||
|
||||
try:
|
||||
import tensorflow as tf
|
||||
if data.dtype == tf.bfloat16:
|
||||
data = tf.convert_to_tensor(data.numpy().astype(np.float32))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
jb_max_cols, jb_max_colwidth, jb_max_rows, jb_float_options = None, None, None, None
|
||||
is_pandas = __is_pandas_can_be_used_for_array(arr)
|
||||
if is_pandas:
|
||||
jb_max_cols, jb_max_colwidth, jb_max_rows, jb_float_options = __set_pd_options(format)
|
||||
|
||||
if is_sort_command:
|
||||
arr['data'] = data
|
||||
data = arr
|
||||
|
||||
format = pd.get_option('display.float_format') if is_pandas else format
|
||||
|
||||
data = fun(data, format)
|
||||
|
||||
if is_pandas:
|
||||
__reset_pd_options(jb_max_cols, jb_max_colwidth, jb_max_rows, jb_float_options)
|
||||
|
||||
return data
|
||||
|
||||
|
||||
def __get_tables_display_options():
|
||||
# type: () -> Tuple[None, Union[int, None], None]
|
||||
try:
|
||||
import pandas as pd
|
||||
# In pandas versions earlier than 1.0, max_colwidth must be set as an integer
|
||||
if int(pd.__version__.split('.')[0]) < 1:
|
||||
return None, MAX_COLWIDTH, None
|
||||
except Exception:
|
||||
pass
|
||||
return None, None, None
|
||||
|
||||
|
||||
def __set_pd_options(format):
|
||||
max_cols, max_colwidth, max_rows = __get_tables_display_options()
|
||||
_jb_float_options = None
|
||||
|
||||
_jb_max_cols = pd.get_option('display.max_columns')
|
||||
_jb_max_colwidth = pd.get_option('display.max_colwidth')
|
||||
_jb_max_rows = pd.get_option('display.max_rows')
|
||||
if format is not None:
|
||||
_jb_float_options = pd.get_option('display.float_format')
|
||||
|
||||
pd.set_option('display.max_columns', max_cols)
|
||||
pd.set_option('display.max_rows', max_rows)
|
||||
try:
|
||||
pd.set_option('display.max_colwidth', max_colwidth)
|
||||
except ValueError:
|
||||
pd.set_option('display.max_colwidth', MAX_COLWIDTH)
|
||||
|
||||
format_function = __define_format_function(format)
|
||||
if format_function is not None:
|
||||
pd.set_option('display.float_format', format_function)
|
||||
|
||||
return _jb_max_cols, _jb_max_colwidth, _jb_max_rows, _jb_float_options
|
||||
|
||||
|
||||
def __reset_pd_options(max_cols, max_colwidth, max_rows, float_format):
|
||||
pd.set_option('display.max_columns', max_cols)
|
||||
pd.set_option('display.max_colwidth', max_colwidth)
|
||||
pd.set_option('display.max_rows', max_rows)
|
||||
if float_format is not None:
|
||||
pd.set_option('display.float_format', float_format)
|
||||
|
||||
|
||||
def __define_format_function(format):
|
||||
# type: (Union[None, str]) -> Union[Callable, None]
|
||||
if format is None or format == 'null':
|
||||
return None
|
||||
|
||||
if type(format) == str and format.startswith("%"):
|
||||
return lambda x: format % x
|
||||
else:
|
||||
return None
|
||||
|
||||
def __is_pandas_can_be_used_for_array(array):
|
||||
is_cpu_stored = True
|
||||
try:
|
||||
device = str(array.device).lower()
|
||||
# check mac
|
||||
if "cpu" in device:
|
||||
is_cpu_stored = True
|
||||
else:
|
||||
is_cpu_stored = False
|
||||
except:
|
||||
pass
|
||||
return is_pd_can_be_imported and is_cpu_stored
|
||||
+520
@@ -0,0 +1,520 @@
|
||||
# Copyright 2000-2023 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import typing
|
||||
from collections import OrderedDict
|
||||
import sys
|
||||
if sys.version_info < (3, 0):
|
||||
from collections import Iterable
|
||||
else:
|
||||
from collections.abc import Iterable
|
||||
|
||||
TABLE_TYPE_NEXT_VALUE_SEPARATOR = '__pydev_table_column_type_val__'
|
||||
MAX_COLWIDTH = 100000
|
||||
CSV_FORMAT_SEPARATOR = '~'
|
||||
DASH_SYMBOL = '\u2014'
|
||||
UNSUPPORTED_KINDS = {"c", "V"} # complex, void/raw
|
||||
OBJECT_SAMPLE_LIMIT = 10
|
||||
|
||||
|
||||
class InspectionResultsDict:
|
||||
KEY_INSPECTION_NAME = "inspection"
|
||||
|
||||
KEY_STATUS = "executionStatus"
|
||||
VALUE_STATUS_SUCCESS = "SUCCESS"
|
||||
VALUE_STATUS_FAILED = "FAILED"
|
||||
|
||||
KEY_IS_TRIGGERED = "isTriggered"
|
||||
VALUE_TRIGGERED_NO = "NO"
|
||||
VALUE_TRIGGERED_YES = "YES"
|
||||
|
||||
KEY_DETAILS = "inspectionResultDetails"
|
||||
KEY_DETAILS_TYPE = "type"
|
||||
VALUE_DETAILS_TYPE_ALL = "All"
|
||||
VALUE_DETAILS_TYPE_PER_COLUMN = "PerColumn"
|
||||
|
||||
KEY_DETAILS_VALUE = "value"
|
||||
|
||||
|
||||
class ColumnVisualisationType:
|
||||
HISTOGRAM = "histogram"
|
||||
UNIQUE = "unique"
|
||||
PERCENTAGE = "percentage"
|
||||
|
||||
|
||||
class ColumnVisualisationUtils:
|
||||
NUM_BINS = 20
|
||||
MAX_UNIQUE_VALUES_TO_SHOW_IN_VIS = 3
|
||||
UNIQUE_VALUES_PERCENT = 50
|
||||
|
||||
TABLE_OCCURRENCES_COUNT_NEXT_COLUMN_SEPARATOR = '__pydev_table_occurrences_count_next_column__'
|
||||
TABLE_OCCURRENCES_COUNT_NEXT_VALUE_SEPARATOR = '__pydev_table_occurrences_count_next_value__'
|
||||
TABLE_OCCURRENCES_COUNT_DICT_SEPARATOR = '__pydev_table_occurrences_count_dict__'
|
||||
TABLE_OCCURRENCES_COUNT_OTHER = '__pydev_table_other__'
|
||||
|
||||
|
||||
def get_type(table):
|
||||
# type: (str) -> str
|
||||
return str(type(table))
|
||||
|
||||
|
||||
# noinspection PyUnresolvedReferences
|
||||
def get_shape(table):
|
||||
# type: (Union[pd.DataFrame, pd.Series]) -> str
|
||||
return str(table.shape[0])
|
||||
|
||||
|
||||
# noinspection PyUnresolvedReferences
|
||||
def get_head(table):
|
||||
# type: (Union[pd.DataFrame, pd.Series]) -> str
|
||||
return repr(__convert_to_df(table).head(1).to_html(notebook=True, max_cols=None))
|
||||
|
||||
|
||||
# noinspection PyUnresolvedReferences
|
||||
def get_column_types(table):
|
||||
# type: (Union[pd.DataFrame, pd.Series]) -> str
|
||||
table = __convert_to_df(table)
|
||||
return str(table.index.dtype) + TABLE_TYPE_NEXT_VALUE_SEPARATOR + \
|
||||
TABLE_TYPE_NEXT_VALUE_SEPARATOR.join([str(t) for t in table.dtypes])
|
||||
|
||||
|
||||
# used by pydevd
|
||||
# noinspection PyUnresolvedReferences
|
||||
def get_data(table, use_csv_serialization, start_index=None, end_index=None, format=None):
|
||||
# type: (Union[pd.DataFrame, pd.Series], bool, int, int) -> str
|
||||
|
||||
def convert_data_to_csv(data, format):
|
||||
return repr(__convert_to_df(data).to_csv(na_rep = "NaN", float_format=format, sep=CSV_FORMAT_SEPARATOR))
|
||||
|
||||
def convert_data_to_html(data, format):
|
||||
return repr(__convert_to_df(data).to_html(notebook=True))
|
||||
|
||||
if use_csv_serialization:
|
||||
computed_data = __compute_sliced_data(table, convert_data_to_csv, start_index, end_index, format)
|
||||
else:
|
||||
computed_data = __compute_sliced_data(table, convert_data_to_html, start_index, end_index, format)
|
||||
return computed_data
|
||||
|
||||
|
||||
# used by DSTableCommands
|
||||
# noinspection PyUnresolvedReferences
|
||||
def display_data_html(table, start_index, end_index):
|
||||
# type: (Union[pd.DataFrame, pd.Series], int, int) -> None
|
||||
def ipython_display(data, format):
|
||||
from IPython.display import display, HTML
|
||||
display(HTML(__convert_to_df(data).to_html(notebook=True)))
|
||||
__compute_sliced_data(table, ipython_display, start_index, end_index)
|
||||
|
||||
|
||||
# used by DSTableCommands
|
||||
# noinspection PyUnresolvedReferences
|
||||
def display_data_csv(table, start_index, end_index):
|
||||
# type: (Union[pd.DataFrame, pd.Series], int, int) -> None
|
||||
def ipython_display(data, format):
|
||||
try:
|
||||
data = data.to_csv(na_rep = "NaN", sep=CSV_FORMAT_SEPARATOR, float_format=format)
|
||||
except AttributeError:
|
||||
pass
|
||||
print(repr(__convert_to_df(data)))
|
||||
__compute_sliced_data(table, ipython_display, start_index, end_index)
|
||||
|
||||
|
||||
def get_column_descriptions(table):
|
||||
# type: (Union[pd.DataFrame, pd.Series]) -> str
|
||||
described_result = __get_describe(table)
|
||||
|
||||
if described_result is not None:
|
||||
return get_data(described_result, None, None)
|
||||
else:
|
||||
return ""
|
||||
|
||||
|
||||
def get_value_occurrences_count(table):
|
||||
import warnings
|
||||
df = __convert_to_df(table)
|
||||
bin_counts = []
|
||||
|
||||
with warnings.catch_warnings():
|
||||
warnings.simplefilter("ignore") # Suppress all
|
||||
for _, column_data in df.items():
|
||||
column_visualisation_type, result = __analyze_column(column_data)
|
||||
|
||||
bin_counts.append(str({column_visualisation_type:result}))
|
||||
return ColumnVisualisationUtils.TABLE_OCCURRENCES_COUNT_NEXT_COLUMN_SEPARATOR.join(bin_counts)
|
||||
|
||||
|
||||
def get_inspection_none_count(table):
|
||||
def _calculate_none_count(cur_table):
|
||||
"""Calculate missing values per column"""
|
||||
results_per_column = []
|
||||
for col, missing_count in cur_table.isna().sum().items():
|
||||
if missing_count > 0:
|
||||
results_per_column.append({
|
||||
"columnName": col,
|
||||
"value": str(missing_count)
|
||||
})
|
||||
|
||||
is_triggered = len(results_per_column) > 0
|
||||
details = __create_per_column_details(results_per_column) if is_triggered else None
|
||||
return __create_success_result(is_triggered, details)
|
||||
|
||||
return __execute_inspection(table, _calculate_none_count, "NONE_COUNT")
|
||||
|
||||
|
||||
def get_inspection_duplicate_rows(table):
|
||||
def _calculate_duplicate_rows(cur_table):
|
||||
duplicate_rows_number = cur_table.duplicated().sum()
|
||||
is_triggered = duplicate_rows_number > 0
|
||||
details = __create_all_details(str(duplicate_rows_number)) if is_triggered else None
|
||||
return __create_success_result(is_triggered, details)
|
||||
|
||||
return __execute_inspection(table, _calculate_duplicate_rows, "DUPLICATE_ROWS")
|
||||
|
||||
|
||||
def get_inspection_outliers(table):
|
||||
def _calculate_outliers(cur_table):
|
||||
results_per_column = []
|
||||
for col in cur_table.columns:
|
||||
if pd.api.types.is_numeric_dtype(cur_table[col]):
|
||||
q1 = cur_table[col].quantile(0.25)
|
||||
q3 = cur_table[col].quantile(0.75)
|
||||
iqr = q3 - q1
|
||||
lower_bound = q1 - 1.5 * iqr
|
||||
upper_bound = q3 + 1.5 * iqr
|
||||
|
||||
# Boolean mask for outliers
|
||||
mask = (cur_table[col] < lower_bound) | (cur_table[col] > upper_bound)
|
||||
outliers_count = cur_table[col][mask].count()
|
||||
|
||||
if outliers_count > 0:
|
||||
results_per_column.append({
|
||||
"columnName": col,
|
||||
"value": str(outliers_count)
|
||||
})
|
||||
|
||||
is_triggered = len(results_per_column) > 0
|
||||
details = __create_per_column_details(results_per_column) if is_triggered else None
|
||||
return __create_success_result(is_triggered, details)
|
||||
|
||||
return __execute_inspection(table, _calculate_outliers, "OUTLIERS")
|
||||
|
||||
|
||||
def get_inspection_constant_columns(table):
|
||||
def _calculate_constant_columns(cur_table):
|
||||
results_per_column = []
|
||||
for col in cur_table.columns:
|
||||
if cur_table[col].nunique(dropna=False) == 1:
|
||||
results_per_column.append({
|
||||
"columnName": col,
|
||||
"value": str(cur_table[col].iloc[0])
|
||||
})
|
||||
|
||||
is_triggered = len(results_per_column) > 0
|
||||
details = __create_per_column_details(results_per_column) if is_triggered else None
|
||||
return __create_success_result(is_triggered, details)
|
||||
|
||||
return __execute_inspection(table, _calculate_constant_columns, "CONSTANT_COLUMNS")
|
||||
|
||||
|
||||
def __get_data_slice(table, start, end):
|
||||
return __convert_to_df(table).iloc[start:end]
|
||||
|
||||
|
||||
def __compute_sliced_data(table, fun, start_index=None, end_index=None, format=None):
|
||||
# type: (Union[pd.DataFrame, pd.Series], function, Union[None, int], Union[None, int], Union[None, str]) -> str
|
||||
|
||||
max_cols, max_colwidth, max_rows = __get_tables_display_options()
|
||||
|
||||
_jb_max_cols = pd.get_option('display.max_columns')
|
||||
_jb_max_colwidth = pd.get_option('display.max_colwidth')
|
||||
_jb_max_rows = pd.get_option('display.max_rows')
|
||||
if format is not None:
|
||||
_jb_float_options = pd.get_option('display.float_format')
|
||||
|
||||
pd.set_option('display.max_columns', max_cols)
|
||||
pd.set_option('display.max_rows', max_rows)
|
||||
pd.set_option('display.max_colwidth', max_colwidth)
|
||||
|
||||
format_function = __define_format_function(format)
|
||||
if format_function is not None:
|
||||
pd.set_option('display.float_format', format_function)
|
||||
|
||||
if start_index is not None and end_index is not None:
|
||||
table = __get_data_slice(table, start_index, end_index)
|
||||
|
||||
data = fun(table, pd.get_option('display.float_format'))
|
||||
|
||||
pd.set_option('display.max_columns', _jb_max_cols)
|
||||
pd.set_option('display.max_colwidth', _jb_max_colwidth)
|
||||
pd.set_option('display.max_rows', _jb_max_rows)
|
||||
if format is not None:
|
||||
pd.set_option('display.float_format', _jb_float_options)
|
||||
|
||||
return data
|
||||
|
||||
|
||||
def __define_format_function(format):
|
||||
# type: (Union[None, str]) -> Union[Callable, None]
|
||||
if format is None or format == 'null':
|
||||
return None
|
||||
|
||||
if type(format) == str and format.startswith("%"):
|
||||
return lambda x: format % x
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def __analyze_column(column):
|
||||
col_type = column.dtype
|
||||
|
||||
if __is_boolean(col_type):
|
||||
return ColumnVisualisationType.HISTOGRAM, __analyze_boolean_column(column)
|
||||
elif __is_categorical(column, col_type):
|
||||
return __analyze_categorical_column(column)
|
||||
elif __is_numeric(col_type):
|
||||
return ColumnVisualisationType.HISTOGRAM, __analyze_numeric_column(column)
|
||||
|
||||
|
||||
def __is_boolean(col_type):
|
||||
return col_type == bool
|
||||
|
||||
|
||||
def __is_categorical(column, col_type):
|
||||
return col_type.kind in ['O', 'S', 'U', 'M', 'm', 'c'] or column.isna().all() or col_type.kind is None
|
||||
|
||||
|
||||
def __is_numeric(col_type):
|
||||
return col_type.kind in ['i', 'f', 'u']
|
||||
|
||||
def __analyze_boolean_column(column):
|
||||
res = column.value_counts().sort_index().to_dict(OrderedDict)
|
||||
return __add_custom_key_value_separator(res.items())
|
||||
|
||||
|
||||
def __analyze_categorical_column(column):
|
||||
# Processing of unhashable types (lists, dicts, etc.).
|
||||
# In Polars these types are NESTED and can be processed separately, but in Pandas they are Objects
|
||||
if len(column) == 0 or not isinstance(column.iloc[0], typing.Hashable):
|
||||
return None, "{}"
|
||||
|
||||
value_counts = column.value_counts(dropna=False, normalize=True, sort=True, ascending=False)
|
||||
all_values = len(column)
|
||||
vis_type = ColumnVisualisationType.PERCENTAGE
|
||||
if len(value_counts) <= 3 or float(len(value_counts)) / all_values * 100 <= ColumnVisualisationUtils.UNIQUE_VALUES_PERCENT:
|
||||
# If column contains <= 3 unique values no `Other` category is shown, but all of these values and their percentages
|
||||
num_unique_values_to_show_in_vis = ColumnVisualisationUtils.MAX_UNIQUE_VALUES_TO_SHOW_IN_VIS - (0 if len(value_counts) == 3 else 1)
|
||||
|
||||
top_values = value_counts.iloc[:num_unique_values_to_show_in_vis].apply(lambda v_c_share: round(v_c_share * 100, 1)).to_dict(OrderedDict)
|
||||
if len(value_counts) == 3:
|
||||
top_values[ColumnVisualisationUtils.TABLE_OCCURRENCES_COUNT_OTHER] = -1
|
||||
else:
|
||||
others_count = value_counts.iloc[num_unique_values_to_show_in_vis:].sum()
|
||||
top_values[ColumnVisualisationUtils.TABLE_OCCURRENCES_COUNT_OTHER] = round(others_count * 100, 1)
|
||||
result = __add_custom_key_value_separator(top_values.items())
|
||||
else:
|
||||
vis_type = ColumnVisualisationType.UNIQUE
|
||||
top_values = len(value_counts)
|
||||
result = top_values
|
||||
return vis_type, result
|
||||
|
||||
|
||||
def __analyze_numeric_column(column):
|
||||
if column.size <= ColumnVisualisationUtils.NUM_BINS:
|
||||
res = column.value_counts().sort_index().to_dict()
|
||||
else:
|
||||
def format_function(x):
|
||||
if x == int(x):
|
||||
return int(x)
|
||||
else:
|
||||
return round(x, 3)
|
||||
|
||||
counts, bin_edges = np.histogram(column.dropna(), bins=ColumnVisualisationUtils.NUM_BINS)
|
||||
|
||||
# so the long dash will be correctly viewed both on Mac and Windows
|
||||
bin_labels = ['{} {} {}'.format(format_function(bin_edges[i]), DASH_SYMBOL, format_function(bin_edges[i+1])) for i in range(ColumnVisualisationUtils.NUM_BINS)]
|
||||
bin_count_dict = {label: count for label, count in zip(bin_labels, counts)}
|
||||
res = bin_count_dict
|
||||
return __add_custom_key_value_separator(res.items())
|
||||
|
||||
|
||||
def __add_custom_key_value_separator(pairs_list):
|
||||
return ColumnVisualisationUtils.TABLE_OCCURRENCES_COUNT_NEXT_VALUE_SEPARATOR.join(
|
||||
['{}{}{}'.format(key, ColumnVisualisationUtils.TABLE_OCCURRENCES_COUNT_DICT_SEPARATOR, value) for key, value in pairs_list]
|
||||
)
|
||||
|
||||
|
||||
# noinspection PyUnresolvedReferences
|
||||
def __convert_to_df(table):
|
||||
# type: (Union[pd.DataFrame, pd.Series, pd.Categorical]) -> pd.DataFrame
|
||||
try:
|
||||
import geopandas
|
||||
if type(table) is geopandas.GeoSeries:
|
||||
return __series_to_df(table)
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
if type(table) is pd.Series:
|
||||
return __series_to_df(table)
|
||||
if type(table) is pd.Categorical:
|
||||
return __categorical_to_df(table)
|
||||
return table
|
||||
|
||||
|
||||
# pandas.Series support
|
||||
def __get_column_name(table):
|
||||
# type: (pd.Series) -> str
|
||||
if table.name is not None:
|
||||
# noinspection PyTypeChecker
|
||||
return table.name
|
||||
return '<unnamed>'
|
||||
|
||||
|
||||
def __series_to_df(table):
|
||||
# type: (pd.Series) -> pd.DataFrame
|
||||
return table.to_frame(name=__get_column_name(table))
|
||||
|
||||
|
||||
# numpy.array support
|
||||
def __array_to_df(table):
|
||||
# type: (np.ndarray) -> pd.DataFrame
|
||||
return pd.DataFrame(table)
|
||||
|
||||
|
||||
def __categorical_to_df(table):
|
||||
# type: (pd.Categorical) -> pd.DataFrame
|
||||
return pd.DataFrame(table)
|
||||
|
||||
|
||||
def __get_tables_display_options():
|
||||
# type: () -> Tuple[None, Union[int, None], None]
|
||||
try:
|
||||
# In pandas versions earlier than 1.0, max_colwidth must be set as an integer
|
||||
if int(pd.__version__.split('.')[0]) < 1:
|
||||
return None, MAX_COLWIDTH, None
|
||||
except ImportError:
|
||||
pass
|
||||
return None, None, None
|
||||
|
||||
|
||||
def __is_iterable(element):
|
||||
# type: (any) -> bool
|
||||
return isinstance(element, Iterable)
|
||||
|
||||
|
||||
def __is_string(element):
|
||||
# type: (any) -> bool
|
||||
return isinstance(element, str)
|
||||
|
||||
|
||||
def __should_skip_describe(element):
|
||||
# type: (any) -> bool
|
||||
if __is_string(element):
|
||||
return False
|
||||
if __is_iterable(element):
|
||||
return True
|
||||
return False
|
||||
|
||||
def __is_summarizable(series):
|
||||
# type: (pd.Series) -> bool
|
||||
kind = series.dtype.kind
|
||||
|
||||
if kind in UNSUPPORTED_KINDS:
|
||||
return False
|
||||
|
||||
# For object dtype, sample some values to check for lists or unstructured types
|
||||
if kind == "O":
|
||||
sample = series.dropna().head(OBJECT_SAMPLE_LIMIT)
|
||||
if sample.map(lambda x: __should_skip_describe(x)).any():
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
def __get_describe_dataframe(table):
|
||||
# type: (pd.DataFrame) -> pd.DataFrame
|
||||
describe_results = []
|
||||
for column_name in table.columns:
|
||||
series = table[column_name]
|
||||
if __is_summarizable(series):
|
||||
describe_results.append(__get_describe_series(series))
|
||||
else:
|
||||
describe_results.append(__get_dummy_describe_series(series))
|
||||
|
||||
return pd.concat(describe_results, axis=1)
|
||||
|
||||
|
||||
def __get_describe_series(series):
|
||||
# type: (pd.Series) -> pd.Series
|
||||
try:
|
||||
return series.describe(percentiles=[.05, .25, .5, .75, .95])
|
||||
except:
|
||||
return __get_dummy_describe_series(series)
|
||||
|
||||
|
||||
def __get_dummy_describe_series(series):
|
||||
# type: (pd.Series) -> pd.Series
|
||||
manual_data = {"count": series.notna().count()}
|
||||
return pd.Series(data = manual_data, index=["count"], name=series.name)
|
||||
|
||||
|
||||
def __get_describe(table):
|
||||
# type: (Union[pd.DataFrame, pd.Series]) -> Union[pd.DataFrame, pd.Series, None]
|
||||
try:
|
||||
if isinstance(table, pd.DataFrame):
|
||||
return __get_describe_dataframe(table)
|
||||
else:
|
||||
if __is_summarizable(table):
|
||||
return __get_describe_series(table)
|
||||
else:
|
||||
return __get_dummy_describe_series(table)
|
||||
except:
|
||||
return None
|
||||
|
||||
|
||||
def __serialize_in_json(result_in_dict, inspection_name):
|
||||
try:
|
||||
import json
|
||||
result_in_dict[InspectionResultsDict.KEY_INSPECTION_NAME] = inspection_name
|
||||
return json.dumps(result_in_dict)
|
||||
except:
|
||||
return '{"%s": "%s" "%s": "%s"}' % (InspectionResultsDict.KEY_INSPECTION_NAME, inspection_name, InspectionResultsDict.KEY_STATUS, InspectionResultsDict.VALUE_STATUS_FAILED)
|
||||
|
||||
|
||||
def __execute_inspection(table, inspection_func, inspection_name):
|
||||
"""Execute inspection with error handling"""
|
||||
try:
|
||||
result = inspection_func(table)
|
||||
return __serialize_in_json(result, inspection_name)
|
||||
except:
|
||||
failed_result = __create_failed_result()
|
||||
return __serialize_in_json(failed_result, inspection_name)
|
||||
|
||||
|
||||
def __create_success_result(is_triggered, details=None):
|
||||
is_triggered_value = InspectionResultsDict.VALUE_TRIGGERED_YES if is_triggered else InspectionResultsDict.VALUE_TRIGGERED_NO
|
||||
result = {
|
||||
InspectionResultsDict.KEY_STATUS: InspectionResultsDict.VALUE_STATUS_SUCCESS,
|
||||
InspectionResultsDict.KEY_IS_TRIGGERED: is_triggered_value
|
||||
}
|
||||
if details:
|
||||
result[InspectionResultsDict.KEY_DETAILS] = details
|
||||
return result
|
||||
|
||||
|
||||
def __create_failed_result():
|
||||
return {
|
||||
InspectionResultsDict.KEY_STATUS: InspectionResultsDict.VALUE_STATUS_FAILED,
|
||||
InspectionResultsDict.KEY_IS_TRIGGERED: InspectionResultsDict.VALUE_TRIGGERED_NO
|
||||
}
|
||||
|
||||
|
||||
def __create_per_column_details(results_per_column):
|
||||
return {
|
||||
InspectionResultsDict.KEY_DETAILS_TYPE: InspectionResultsDict.VALUE_DETAILS_TYPE_PER_COLUMN,
|
||||
InspectionResultsDict.KEY_DETAILS_VALUE: results_per_column
|
||||
}
|
||||
|
||||
def __create_all_details(value):
|
||||
return {
|
||||
InspectionResultsDict.KEY_DETAILS_TYPE: InspectionResultsDict.VALUE_DETAILS_TYPE_ALL,
|
||||
InspectionResultsDict.KEY_DETAILS_VALUE: value
|
||||
}
|
||||
+293
@@ -0,0 +1,293 @@
|
||||
# Copyright 2000-2023 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
import polars as pl
|
||||
|
||||
TABLE_TYPE_NEXT_VALUE_SEPARATOR = '__pydev_table_column_type_val__'
|
||||
MAX_COLWIDTH = 100000
|
||||
pl_version_major, pl_version_minor, _ = pl.__version__.split(".")
|
||||
pl_version_major, pl_version_minor = int(pl_version_major), int(pl_version_minor)
|
||||
COUNT_COL_NAME = "counts" if pl_version_major == 0 and pl_version_minor < 20 else "count"
|
||||
CSV_FORMAT_SEPARATOR = '~'
|
||||
DASH_SYMBOL = '\u2014'
|
||||
|
||||
|
||||
class ColumnVisualisationType:
|
||||
HISTOGRAM = "histogram"
|
||||
UNIQUE = "unique"
|
||||
PERCENTAGE = "percentage"
|
||||
|
||||
|
||||
class ColumnVisualisationUtils:
|
||||
NUM_BINS = 20
|
||||
MAX_UNIQUE_VALUES = 3
|
||||
MAX_UNIQUE_VALUES_TO_SHOW_IN_VIS = 50
|
||||
MAX_VALUES_LENGTH = 100
|
||||
|
||||
TABLE_OCCURRENCES_COUNT_NEXT_COLUMN_SEPARATOR = '__pydev_table_occurrences_count_next_column__'
|
||||
TABLE_OCCURRENCES_COUNT_NEXT_VALUE_SEPARATOR = '__pydev_table_occurrences_count_next_value__'
|
||||
TABLE_OCCURRENCES_COUNT_DICT_SEPARATOR = '__pydev_table_occurrences_count_dict__'
|
||||
TABLE_OCCURRENCES_COUNT_OTHER = '__pydev_table_other__'
|
||||
|
||||
|
||||
def get_type(table):
|
||||
# type: (Union[pl.Series, pl.DataFrame]) -> str
|
||||
return str(type(table))
|
||||
|
||||
|
||||
def get_shape(table):
|
||||
# type: (Union[pl.Series, pl.DataFrame]) -> str
|
||||
return str(table.shape)
|
||||
|
||||
|
||||
def get_head(table):
|
||||
# type: (Union[pl.Series, pl.DataFrame]) -> str
|
||||
with __create_config():
|
||||
return table.head(1)._repr_html_()
|
||||
|
||||
|
||||
def get_column_types(table):
|
||||
# type: (Union[pl.Series, pl.DataFrame]) -> str
|
||||
if type(table) == pl.Series:
|
||||
return str(table.dtype)
|
||||
else:
|
||||
return TABLE_TYPE_NEXT_VALUE_SEPARATOR.join([str(t) for t in table.dtypes])
|
||||
|
||||
|
||||
# used by pydevd
|
||||
def get_data(table, use_csv_serialization, start_index=None, end_index=None, format=None):
|
||||
# type: (Union[pl.Series, pl.DataFrame], int, int) -> str
|
||||
with __create_config(format):
|
||||
if use_csv_serialization:
|
||||
float_precision = __get_float_precision(format)
|
||||
return __write_to_csv(__get_df_slice(table, start_index, end_index), float_precision=float_precision)
|
||||
return table[start_index:end_index]._repr_html_()
|
||||
|
||||
|
||||
# used by DSTableCommands
|
||||
def display_data_html(table, start_index, end_index):
|
||||
# type: (Union[pl.Series, pl.DataFrame], int, int) -> None
|
||||
with __create_config():
|
||||
print(table[start_index:end_index]._repr_html_())
|
||||
|
||||
|
||||
def display_data_csv(table, start_index, end_index):
|
||||
# type: (Union[pl.Series, pl.DataFrame], int, int) -> None
|
||||
with __create_config():
|
||||
print(repr(__write_to_csv(__get_df_slice(table, start_index, end_index))))
|
||||
|
||||
|
||||
def get_column_descriptions(table):
|
||||
# type: (Union[pl.Series, pl.DataFrame]) -> str
|
||||
described_results = __get_describe(table)
|
||||
|
||||
if described_results is not None:
|
||||
return get_data(described_results, None, None)
|
||||
else:
|
||||
return ""
|
||||
|
||||
|
||||
def get_value_occurrences_count(table):
|
||||
# type: (Union[pl.Series, pl.DataFrame]) -> str
|
||||
bin_counts = []
|
||||
if type(table) == pl.DataFrame:
|
||||
for col in table.columns:
|
||||
bin_counts.append(__analyze_column(col, table[col]))
|
||||
elif type(table) == pl.Series:
|
||||
bin_counts.append(__analyze_column(table.name, table))
|
||||
|
||||
return ColumnVisualisationUtils.TABLE_OCCURRENCES_COUNT_NEXT_COLUMN_SEPARATOR.join(bin_counts)
|
||||
|
||||
|
||||
def __analyze_column(col_name, column):
|
||||
col_type = column.dtype
|
||||
res = []
|
||||
column_visualisation_type = None
|
||||
|
||||
if __is_boolean(column, col_type):
|
||||
column_visualisation_type, res = __analyze_boolean_column(column, col_name)
|
||||
|
||||
elif __is_numeric(column, col_type):
|
||||
column_visualisation_type, res = __analyze_numeric_column(column, col_name)
|
||||
|
||||
else:
|
||||
column_visualisation_type, res = __analyze_categorical_column(column, col_name)
|
||||
|
||||
if column_visualisation_type != ColumnVisualisationType.UNIQUE:
|
||||
counts = ColumnVisualisationUtils.TABLE_OCCURRENCES_COUNT_NEXT_VALUE_SEPARATOR.join(res)
|
||||
else:
|
||||
counts = res
|
||||
|
||||
return str({column_visualisation_type:counts})
|
||||
|
||||
|
||||
def __is_boolean(column, col_type):
|
||||
# (pl.Series, pl.DataType) -> bool
|
||||
return col_type == pl.Boolean and not column.is_null().any()
|
||||
|
||||
|
||||
def __is_numeric(column, col_type):
|
||||
# (pl.Series, pl.DataType) -> bool
|
||||
return __is_series_numeric(column) and not column.is_null().all()
|
||||
|
||||
|
||||
def __analyze_boolean_column(column, col_name):
|
||||
counts = column.value_counts().sort(by=col_name).to_dict()
|
||||
return ColumnVisualisationType.HISTOGRAM, __add_custom_key_value_separator(zip(counts[col_name], counts[COUNT_COL_NAME]))
|
||||
|
||||
|
||||
def __analyze_categorical_column(column, col_name):
|
||||
all_values = column.shape[0]
|
||||
if column.is_null().all():
|
||||
value_counts = pl.DataFrame({col_name: "None", COUNT_COL_NAME: all_values})
|
||||
else:
|
||||
value_counts = column.value_counts()
|
||||
|
||||
# Sort in descending order to get values with max percent
|
||||
value_counts = value_counts.sort(COUNT_COL_NAME).reverse()
|
||||
|
||||
if len(value_counts) <= ColumnVisualisationUtils.MAX_UNIQUE_VALUES or len(value_counts) / all_values * 100 <= ColumnVisualisationUtils.MAX_UNIQUE_VALUES_TO_SHOW_IN_VIS:
|
||||
column_visualisation_type = ColumnVisualisationType.PERCENTAGE
|
||||
|
||||
# If column contains <= 3 unique values no `Other` category is shown, but all of these values and their percentages
|
||||
num_unique_values_to_show_in_vis = ColumnVisualisationUtils.MAX_UNIQUE_VALUES - (0 if len(value_counts) == 3 else 1)
|
||||
counts = value_counts[:num_unique_values_to_show_in_vis]
|
||||
counts = counts.with_columns(((pl.col(COUNT_COL_NAME) / all_values * 100).round(1)).alias(COUNT_COL_NAME))
|
||||
top_values = {}
|
||||
for label, count in zip(counts[col_name], counts[COUNT_COL_NAME]):
|
||||
# we should process separately a case with dtype == pl.List
|
||||
if type(label) == pl.Series or column.dtype == pl.List:
|
||||
label_values = label.to_list()
|
||||
label_values_in_str = str(label_values)
|
||||
top_values[label_values_in_str[:ColumnVisualisationUtils.MAX_VALUES_LENGTH]] = count
|
||||
else:
|
||||
label_in_str = str(label)
|
||||
top_values[label_in_str[:ColumnVisualisationUtils.MAX_VALUES_LENGTH]] = count
|
||||
if len(value_counts) == ColumnVisualisationUtils.MAX_UNIQUE_VALUES:
|
||||
top_values[ColumnVisualisationUtils.TABLE_OCCURRENCES_COUNT_OTHER] = -1
|
||||
else:
|
||||
others_count = value_counts[num_unique_values_to_show_in_vis:][COUNT_COL_NAME].sum()
|
||||
top_values[ColumnVisualisationUtils.TABLE_OCCURRENCES_COUNT_OTHER] = round(others_count / all_values * 100, 1)
|
||||
res = __add_custom_key_value_separator(top_values.items())
|
||||
|
||||
else:
|
||||
column_visualisation_type = ColumnVisualisationType.UNIQUE
|
||||
res = len(value_counts)
|
||||
return column_visualisation_type, res
|
||||
|
||||
|
||||
def __analyze_numeric_column(column, col_name):
|
||||
# handle np.NaN values, because they are not dropped with drop_nulls() the way they are
|
||||
column = column.fill_null(strategy="min")
|
||||
if column.shape[0] <= ColumnVisualisationUtils.NUM_BINS:
|
||||
raw_counts = column.value_counts().sort(by=col_name).to_dict(as_series=False)
|
||||
res = __add_custom_key_value_separator(zip(raw_counts[col_name], raw_counts[COUNT_COL_NAME]))
|
||||
else:
|
||||
import numpy as np
|
||||
|
||||
def format_function(x):
|
||||
if x == int(x):
|
||||
return int(x)
|
||||
else:
|
||||
return round(x, 3)
|
||||
|
||||
counts, bin_edges = np.histogram(column, bins=ColumnVisualisationUtils.NUM_BINS)
|
||||
# so the long dash will be correctly viewed both on Mac and Windows
|
||||
bin_labels = ['{} {} {}'.format(format_function(bin_edges[i]), DASH_SYMBOL, format_function(bin_edges[i + 1])) for i in range(ColumnVisualisationUtils.NUM_BINS)]
|
||||
res = __add_custom_key_value_separator(zip(bin_labels, counts))
|
||||
|
||||
return ColumnVisualisationType.HISTOGRAM, res
|
||||
|
||||
|
||||
def __get_df_slice(table, start_index, end_index):
|
||||
# type: (Union[pl.Series, pl.DataFrame], int, int) -> pl.DataFrame
|
||||
if type(table) == pl.Series:
|
||||
return table[start_index:end_index].to_frame()
|
||||
return table[start_index:end_index]
|
||||
|
||||
|
||||
def __write_to_csv(table, null_value="null", float_precision=None):
|
||||
def serialize_nested(value, null_value="null", float_precision=None):
|
||||
if value is None:
|
||||
return null_value
|
||||
elif isinstance(value, float) and float_precision is not None:
|
||||
return "{:.{}f}".format(value, float_precision)
|
||||
elif isinstance(value, dict):
|
||||
return "{" + ", ".join("{}: {}".format(k, serialize_nested(v, null_value, float_precision)) for k, v in value.items()) + "}"
|
||||
elif isinstance(value, list):
|
||||
return "[" + ", ".join(serialize_nested(v, null_value, float_precision) for v in value) + "]"
|
||||
else:
|
||||
return str(value)
|
||||
|
||||
lines = []
|
||||
lines.append(CSV_FORMAT_SEPARATOR.join(table.columns))
|
||||
for row in table.rows():
|
||||
line = []
|
||||
for value in row:
|
||||
line.append(serialize_nested(value, null_value, float_precision))
|
||||
lines.append(CSV_FORMAT_SEPARATOR.join(line))
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def __create_config(format=None):
|
||||
# type: (Union[str, None]) -> pl.Config
|
||||
cfg = pl.Config()
|
||||
cfg.set_tbl_cols(-1) # Unlimited
|
||||
cfg.set_tbl_rows(-1) # Unlimited
|
||||
cfg.set_fmt_str_lengths(MAX_COLWIDTH) # No option to set unlimited, so it's 100_000
|
||||
float_precision = __get_float_precision(format)
|
||||
if float_precision is not None and hasattr(cfg, 'set_float_precision'):
|
||||
cfg.set_float_precision(float_precision)
|
||||
return cfg
|
||||
|
||||
|
||||
def __add_custom_key_value_separator(pairs_list):
|
||||
return [str(label) + ColumnVisualisationUtils.TABLE_OCCURRENCES_COUNT_DICT_SEPARATOR + str(count) for label, count in pairs_list]
|
||||
|
||||
|
||||
def __get_describe(table):
|
||||
# type: (Union[pl.Series, pl.DataFrame]) -> Union[pl.DataFrame, None]
|
||||
try:
|
||||
if type(table) == pl.DataFrame and 'describe' in table.columns:
|
||||
import random
|
||||
random_suffix = ''.join([chr(random.randint(97, 122)) for _ in range(5)])
|
||||
described_df = table\
|
||||
.rename({'describe': 'describe_original_' + random_suffix})\
|
||||
.describe(percentiles=(0.05, 0.25, 0.5, 0.75, 0.95))
|
||||
else:
|
||||
described_df = table.describe(percentiles=(0.05, 0.25, 0.5, 0.75, 0.95))
|
||||
return described_df
|
||||
# If DataFrame/Series have unsupported type for describe
|
||||
# then Polars will raise TypeError exception. We should catch them.
|
||||
except Exception as e:
|
||||
return
|
||||
|
||||
|
||||
def __get_float_precision(format):
|
||||
# type: (Union[str, None]) -> Union[int, None]
|
||||
if isinstance(format, str):
|
||||
if format.startswith("%") and format.endswith("f"):
|
||||
start = format.find('%.') + 2
|
||||
end = format.find('f')
|
||||
|
||||
if start < end:
|
||||
try:
|
||||
precision = int(format[start:end])
|
||||
return precision
|
||||
except:
|
||||
pass
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def __is_series_numeric(column):
|
||||
"""
|
||||
Determines if the given column is numeric based on the version of the polars
|
||||
library being used.
|
||||
|
||||
For polars major version 0, the method checks if the column is numeric using
|
||||
the `is_numeric` method directly on the column. For later versions, it checks
|
||||
the `is_numeric` method on the column's data type.
|
||||
"""
|
||||
if pl_version_major == 0:
|
||||
return column.is_numeric()
|
||||
else:
|
||||
return column.dtype.is_numeric()
|
||||
@@ -25,6 +25,7 @@ from _pydevd_bundle.pydevd_comm import (
|
||||
internal_get_exception_details_json,
|
||||
internal_step_in_thread,
|
||||
internal_smart_step_into,
|
||||
InternalTableCommand
|
||||
)
|
||||
from _pydevd_bundle.pydevd_comm_constants import (
|
||||
CMD_THREAD_SUSPEND,
|
||||
@@ -341,6 +342,10 @@ class PyDevdAPI(object):
|
||||
int_cmd = InternalGetArray(seq, roffset, coffset, rows, cols, fmt, thread_id, frame_id, scope, attrs)
|
||||
py_db.post_internal_command(int_cmd, thread_id)
|
||||
|
||||
def request_get_table(self, py_db, seq, thread_id, frame_id, init_command, command_type, start_index, end_index, format):
|
||||
int_cmd = InternalTableCommand(seq, thread_id, frame_id, init_command, command_type, start_index, end_index, format)
|
||||
py_db.post_internal_command(int_cmd, thread_id)
|
||||
|
||||
def request_load_full_value(self, py_db, seq, thread_id, frame_id, vars):
|
||||
int_cmd = InternalLoadFullValue(seq, thread_id, frame_id, vars)
|
||||
py_db.post_internal_command(int_cmd, thread_id)
|
||||
|
||||
@@ -95,6 +95,7 @@ from _pydevd_bundle._debug_adapter.pydevd_schema import (
|
||||
)
|
||||
from _pydevd_bundle._debug_adapter import pydevd_base_schema, pydevd_schema
|
||||
from _pydevd_bundle.pydevd_net_command import NetCommand
|
||||
from _pydevd_bundle.custom.pydevd_tables import exec_table_command
|
||||
from _pydevd_bundle.pydevd_xml import ExceptionOnEvaluate
|
||||
from _pydevd_bundle.pydevd_constants import ForkSafeLock, NULL
|
||||
from _pydevd_bundle.pydevd_daemon_thread import PyDBDaemonThread
|
||||
@@ -132,6 +133,7 @@ from io import StringIO
|
||||
|
||||
# CMD_XXX constants imported for backward compatibility
|
||||
from _pydevd_bundle.pydevd_comm_constants import * # @UnusedWildImport
|
||||
from _pydevd_bundle.custom import pydevd_vars as pydevd_custom_vars
|
||||
|
||||
# Socket import aliases:
|
||||
AF_INET, AF_INET6, SOCK_STREAM, SHUT_WR, SOL_SOCKET, IPPROTO_TCP, socket = (
|
||||
@@ -875,7 +877,7 @@ class InternalGetArray(InternalThreadCommand):
|
||||
try:
|
||||
frame = dbg.find_frame(self.thread_id, self.frame_id)
|
||||
var = pydevd_vars.eval_in_context(self.name, frame.f_globals, frame.f_locals, py_db=dbg)
|
||||
xml = pydevd_vars.table_like_struct_to_xml(var, self.name, self.roffset, self.coffset, self.rows, self.cols, self.format)
|
||||
xml = pydevd_custom_vars.table_like_struct_to_xml(var, self.name, self.roffset, self.coffset, self.rows, self.cols, self.format)
|
||||
cmd = dbg.cmd_factory.make_get_array_message(self.sequence, xml)
|
||||
dbg.writer.add_command(cmd)
|
||||
except:
|
||||
@@ -1928,3 +1930,43 @@ class GetValueAsyncThreadConsole(AbstractGetValueAsyncThread):
|
||||
def send_result(self, xml):
|
||||
if self.frame_accessor is not None:
|
||||
self.frame_accessor.ReturnFullValue(self.seq, xml.getvalue())
|
||||
|
||||
#=======================================================================================================================
|
||||
# InternalDataViewerAction
|
||||
#=======================================================================================================================
|
||||
class InternalTableCommand(InternalThreadCommand):
|
||||
def __init__(self, sequence, thread_id, frame_id, init_command, command_type,
|
||||
start_index, end_index, format):
|
||||
InternalThreadCommand.__init__(self, thread_id)
|
||||
self.sequence = sequence
|
||||
self.frame_id = frame_id
|
||||
self.init_command = init_command
|
||||
self.command_type = command_type
|
||||
self.start_index = start_index
|
||||
self.end_index = end_index
|
||||
self.format = format
|
||||
|
||||
def do_it(self, dbg):
|
||||
try:
|
||||
pydev_log.info(f"WE ARE IN INTERNAL TABLE COMMAND, thread_id: {self.thread_id}, frame_id: {self.frame_id}" )
|
||||
frame = dbg.find_frame(self.thread_id, self.frame_id)
|
||||
pydev_log.info("frame = dbg.find_frame(self.thread_id, self.frame_id)")
|
||||
pydev_log.info(f"frame {frame}")
|
||||
|
||||
success, res = self.exec_command(frame)
|
||||
if success:
|
||||
pydev_log.info("success")
|
||||
cmd = dbg.cmd_factory.make_get_table_message(self.sequence, res)
|
||||
dbg.writer.add_command(cmd)
|
||||
else:
|
||||
pydev_log.info(f"error, no success, res: {res}")
|
||||
cmd = dbg.cmd_factory.make_error_message(self.sequence, str(res))
|
||||
dbg.writer.add_command(cmd)
|
||||
except Exception as e:
|
||||
cmd = dbg.cmd_factory.make_error_message(self.sequence, get_exception_traceback_str())
|
||||
dbg.writer.add_command(cmd)
|
||||
|
||||
def exec_command(self, frame):
|
||||
return exec_table_command(self.init_command, self.command_type,
|
||||
self.start_index, self.end_index, self.format,
|
||||
frame.f_globals, frame.f_locals)
|
||||
|
||||
@@ -98,6 +98,10 @@ CMD_LOAD_SOURCE_FROM_FRAME_ID = 207
|
||||
|
||||
CMD_SET_FUNCTION_BREAK = 208
|
||||
|
||||
# Powerful DataViewer commands
|
||||
CMD_DATAVIEWER_ACTION = 210
|
||||
CMD_TABLE_EXEC = 211
|
||||
|
||||
CMD_VERSION = 501
|
||||
CMD_RETURN = 502
|
||||
CMD_SET_PROTOCOL = 503
|
||||
@@ -186,6 +190,7 @@ ID_TO_MEANING = {
|
||||
"205": "CMD_AUTHENTICATE",
|
||||
"206": "CMD_STEP_INTO_COROUTINE",
|
||||
"207": "CMD_LOAD_SOURCE_FROM_FRAME_ID",
|
||||
"211": "CMD_TABLE_EXEC",
|
||||
"501": "CMD_VERSION",
|
||||
"502": "CMD_RETURN",
|
||||
"503": "CMD_SET_PROTOCOL",
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
# Copyright 2000-2025 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
|
||||
from _pydevd_bundle.pydevd_constants import get_current_thread_id, Null, ForkSafeLock
|
||||
from pydevd_file_utils import get_abs_path_real_path_and_base_from_frame
|
||||
from _pydev_bundle._pydev_saved_modules import thread, threading
|
||||
|
||||
+32
-1
@@ -7,7 +7,7 @@ import socket as socket_module
|
||||
from _pydev_bundle._pydev_imports_tipper import TYPE_IMPORT, TYPE_CLASS, TYPE_FUNCTION, TYPE_ATTR, TYPE_BUILTIN, TYPE_PARAM
|
||||
from _pydev_bundle.pydev_is_thread_alive import is_thread_alive
|
||||
from _pydev_bundle.pydev_override import overrides
|
||||
from _pydevd_bundle._debug_adapter import pydevd_schema
|
||||
from _pydevd_bundle._debug_adapter import pydevd_schema, pydevd_base_schema
|
||||
from _pydevd_bundle._debug_adapter.pydevd_schema import (
|
||||
ModuleEvent,
|
||||
ModuleEventBody,
|
||||
@@ -56,6 +56,13 @@ import linecache
|
||||
from io import StringIO
|
||||
from _pydev_bundle import pydev_log
|
||||
|
||||
from _pydevd_bundle._debug_adapter.pydevd_schema import \
|
||||
GetTableResponseBody
|
||||
from _pydevd_bundle.pydevd_comm_constants import CMD_TABLE_EXEC
|
||||
|
||||
from _pydevd_bundle._debug_adapter.pydevd_schema import \
|
||||
GetArrayResponseBody
|
||||
|
||||
|
||||
class ModulesManager(object):
|
||||
def __init__(self):
|
||||
@@ -584,3 +591,27 @@ This may mean a number of things:
|
||||
def make_exit_command(self, py_db):
|
||||
event = pydevd_schema.TerminatedEvent(pydevd_schema.TerminatedEventBody())
|
||||
return NetCommand(CMD_EXIT, 0, event, is_json=True)
|
||||
|
||||
@overrides(NetCommandFactory.make_get_table_message)
|
||||
def make_get_table_message(self, seq, res):
|
||||
try:
|
||||
body = GetTableResponseBody(result=res)
|
||||
pydev_log.info(f"RESPONSE BODY: {body}")
|
||||
response = pydevd_schema.GetTableResponse(request_seq=seq, success=True, body=body)
|
||||
pydev_log.info(f"RESPONSE: {response}")
|
||||
return NetCommand(CMD_RETURN, 0, response, is_json=True)
|
||||
except Exception as e:
|
||||
pydev_log.exception(f"Error while building getTable response: {e}")
|
||||
err_response = pydevd_schema.GetTableResponse(request_seq=seq, success=False, body={})
|
||||
return NetCommand(CMD_RETURN, 0, err_response, is_json=True)
|
||||
|
||||
@overrides(NetCommandFactory.make_get_array_message)
|
||||
def make_get_array_message(self, seq, res):
|
||||
try:
|
||||
body = GetArrayResponseBody(result=res)
|
||||
response = pydevd_schema.GetArrayResponse(request_seq=seq, success=True, body=body)
|
||||
return NetCommand(CMD_RETURN, 0, response, is_json=True)
|
||||
except Exception as e:
|
||||
pydev_log.exception(f"Error while building getTable response: {e}")
|
||||
err_response = pydevd_schema.GetArrayResponse(request_seq=seq, success=False, body={})
|
||||
return NetCommand(CMD_RETURN, 0, err_response, is_json=True)
|
||||
+5
@@ -42,6 +42,7 @@ from _pydevd_bundle.pydevd_comm_constants import (
|
||||
VERSION_STRING,
|
||||
CMD_RELOAD_CODE,
|
||||
CMD_LOAD_SOURCE_FROM_FRAME_ID,
|
||||
CMD_TABLE_EXEC,
|
||||
)
|
||||
from _pydevd_bundle.pydevd_constants import (
|
||||
DebugInfoHolder,
|
||||
@@ -556,3 +557,7 @@ This may mean a number of things:
|
||||
|
||||
def make_exit_command(self, py_db):
|
||||
return NULL_EXIT_COMMAND
|
||||
|
||||
def make_get_table_message(self, seq, res):
|
||||
res_xml = "<xml>" + res + "</xml>"
|
||||
return NetCommand(CMD_TABLE_EXEC, seq, res_xml)
|
||||
@@ -231,6 +231,22 @@ class _PyDevCommandProcessor(object):
|
||||
|
||||
self.api.request_get_array(py_db, seq, roffset, coffset, rows, cols, format, thread_id, frame_id, scope, attrs)
|
||||
|
||||
def cmd_table_exec(self, py_db, cmd_id, seq, text):
|
||||
try:
|
||||
parameters = text.split('\t')
|
||||
thread_id, frame_id, init_command, command_type = parameters[:4]
|
||||
|
||||
start_index, end_index, format = None, None, None
|
||||
|
||||
if len(parameters) >= 7:
|
||||
start_index = int(parameters[4])
|
||||
end_index = int(parameters[5])
|
||||
format = parameters[6]
|
||||
|
||||
self.api.request_get_table(py_db, seq, thread_id, frame_id, init_command, command_type, start_index, end_index, format)
|
||||
except:
|
||||
traceback.print_exc()
|
||||
|
||||
def cmd_show_return_values(self, py_db, cmd_id, seq, text):
|
||||
show_return_values = text.split("\t")[1]
|
||||
self.api.set_show_return_values(py_db, int(show_return_values) == 1)
|
||||
|
||||
+47
@@ -1352,3 +1352,50 @@ class PyDevJsonCommandProcessor(object):
|
||||
|
||||
response = pydevd_base_schema.build_response(request)
|
||||
return NetCommand(CMD_RETURN, 0, response, is_json=True)
|
||||
|
||||
def on_gettable_request(self, py_db, request):
|
||||
args = request.arguments
|
||||
thread_id = args.threadId
|
||||
frame_id = args.frameId
|
||||
init_command = args.command
|
||||
command_type = args.commandType
|
||||
start_index = args.start
|
||||
end_index = args.end
|
||||
df_format = args.format
|
||||
error_msg = self.api.request_get_table(py_db, request.seq, thread_id, frame_id, init_command, command_type, start_index, end_index, df_format)
|
||||
if error_msg:
|
||||
response = pydevd_base_schema.build_response(
|
||||
request,
|
||||
kwargs={
|
||||
"body": {},
|
||||
"success": False,
|
||||
"message": error_msg,
|
||||
},
|
||||
)
|
||||
pydev_log.error("ERR WHILE EXECUTING GETTABLE" + error_msg)
|
||||
return NetCommand(CMD_RETURN, 0, response, is_json=True)
|
||||
return None
|
||||
|
||||
def on_getarray_request(self, py_db, request):
|
||||
args = request.arguments
|
||||
thread_id = args.threadId
|
||||
frame_id = args.frameId
|
||||
row_offset = args.rowOffset
|
||||
col_offset = args.colOffset
|
||||
rows = args.rows
|
||||
cols = args.cols
|
||||
fmt = args.format
|
||||
attrs = args.variableName
|
||||
error_msg = self.api.request_get_array(py_db, request.seq, row_offset, col_offset, rows, cols, fmt, thread_id, frame_id, None, attrs)
|
||||
if error_msg:
|
||||
response = pydevd_base_schema.build_response(
|
||||
request,
|
||||
kwargs={
|
||||
"body": {},
|
||||
"success": False,
|
||||
"message": error_msg,
|
||||
},
|
||||
)
|
||||
pydev_log.error("ERR WHILE EXECUTING GETTABLE" + error_msg)
|
||||
return NetCommand(CMD_RETURN, 0, response, is_json=True)
|
||||
return None
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
from __future__ import nested_scopes
|
||||
import traceback
|
||||
import warnings
|
||||
from _pydev_bundle import pydev_log
|
||||
from _pydev_bundle._pydev_saved_modules import thread, threading
|
||||
from _pydev_bundle import _pydev_saved_modules
|
||||
@@ -22,6 +21,7 @@ from _pydevd_bundle.pydevd_constants import (
|
||||
PYDEVD_WARN_SLOW_RESOLVE_TIMEOUT,
|
||||
get_global_debugger,
|
||||
)
|
||||
from _pydevd_bundle.custom.pydevd_asyncio_provider import get_eval_async_expression_in_context
|
||||
|
||||
|
||||
def save_main_module(file, module_name):
|
||||
|
||||
Reference in New Issue
Block a user