Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 21 additions & 0 deletions objex/explorer.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,9 @@
from .schema import _INDICES
from .dbutils import _run_ddl

# display cap for repr snippets appended to object labels
_LABEL_REPR_LIMIT = 40


class InvalidDatabaseError(ValueError):
pass
Expand Down Expand Up @@ -1040,6 +1043,10 @@ def object_label(self, obj_id):
name = 'frame ' + self.frame_codename(obj_id)
else:
name = self.obj_typequalname(obj_id)
value = self.obj_repr(obj_id)
if value is not None:
name = '{} {}'.format(
name, value if len(value) <= _LABEL_REPR_LIMIT else value[:_LABEL_REPR_LIMIT - 1] + '…')
return '<{}{}#{}>'.format(name, mark_label, obj_id)

def obj_attributed_size(self, obj_id):
Expand All @@ -1051,6 +1058,15 @@ def obj_attributed_size(self, obj_id):
default=self.obj_size(obj_id),
)

def obj_repr(self, obj_id):
if 'object_repr' not in self._table_names:
return None
return self.sql_val(
'SELECT repr FROM object_repr WHERE object = ?',
(obj_id,),
default=None,
)

def object_summary(self, obj_id):
refcount = self.obj_refcount(obj_id)
size = self.obj_size(obj_id)
Expand All @@ -1066,6 +1082,7 @@ def object_summary(self, obj_id):
'refcount': refcount,
'refcount_display': self.format_refcount(refcount),
'len': self.obj_len(obj_id),
'repr': self.obj_repr(obj_id),
'marks': self.get_marks(obj_id),
'flags': {
'is_type': bool(self.obj_is_type(obj_id)),
Expand Down Expand Up @@ -1380,6 +1397,10 @@ def _obj_label(self, obj_id):
name = "frame " + self.reader.frame_codename(obj_id)
else:
name = self.reader.obj_typequalname(obj_id)
value = self.reader.obj_repr(obj_id)
if value is not None:
name = "{} {}".format(
name, value if len(value) <= _LABEL_REPR_LIMIT else value[:_LABEL_REPR_LIMIT - 1] + '…')
coloring = lambda s: colored(s, 'red')
return coloring("<{}{}#{}>".format(name, mark_label, obj_id))

Expand Down
167 changes: 157 additions & 10 deletions objex/exporter.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
import resource
except ImportError: # windows
resource = None
import reprlib
import shutil
import sys
from socket import getfqdn
Expand Down Expand Up @@ -87,9 +88,17 @@ class _Writer:
_TRACKED_TYPES = (types.ModuleType, types.FrameType, types.FunctionType, types.CodeType)
_COMMIT_INTERVAL_OBJECTS = 10000

def __init__(self, conn, use_gc=False):
def __init__(self, conn, use_gc=False, values='none', value_limit=128, redact=True):
if values not in ('none', 'builtins', 'repr'):
raise ValueError("values must be one of 'none', 'builtins', 'repr'")
self.conn = conn
self.use_gc = use_gc
self.values = values
self.value_limit = value_limit
self.redact = redact
self._reprlib = reprlib.Repr()
self._reprlib.maxstring = value_limit
self._reprlib.maxother = value_limit
gc.collect() # try to minimize garbage
# track these to separate out "extra" objects that are generated as part of
# the object walking process
Expand Down Expand Up @@ -124,7 +133,7 @@ def __init__(self, conn, use_gc=False):
# self.times = []

@classmethod
def write_to_path(cls, path, use_gc=False, use_wal=True):
def write_to_path(cls, path, use_gc=False, use_wal=True, values='none', value_limit=128, redact=True):
'''create a new instance that will dump state to path (which shouldn't exist)'''
conn = sqlite3.connect(path)
conn.text_factory = str
Expand All @@ -136,12 +145,14 @@ def write_to_path(cls, path, use_gc=False, use_wal=True):
num_collected = _gc_prep()
conn.execute(
"""
INSERT INTO meta (id, pid, hostname, memory_mb, gc_info, num_gcd_objects)
VALUES (0, ?, ?, ?, ?, ?)
INSERT INTO meta (id, pid, hostname, memory_mb, gc_info, num_gcd_objects, capture_values)
VALUES (0, ?, ?, ?, ?, ?, ?)
""",
(os.getpid(), getfqdn(), memory, '[{},{},{}]'.format(*gc.get_count()), num_collected))
writer = cls(conn, use_gc=use_gc)
(os.getpid(), getfqdn(), memory, '[{},{},{}]'.format(*gc.get_count()), num_collected, values))
writer = cls(conn, use_gc=use_gc, values=values, value_limit=value_limit, redact=redact)
writer.add_all()
if writer.values != 'none' and writer.redact:
writer.redact_sensitive_values()
writer.finish()
except Exception:
conn.close()
Expand Down Expand Up @@ -199,6 +210,8 @@ def _ensure_db_id(self, obj, is_type=False, refs=0):
in_gc_objects,
is_gc_tracked)
)
if self.values != 'none':
self._maybe_capture_value(obj, obj_id)
# all of these are pretty rare (maybe optimize?)
if id(obj) not in self.type_id_map and (is_type or isinstance(obj, type)):
obj_type_id = self.type_id_map[id(obj)] = len(self.type_id_map)
Expand All @@ -216,6 +229,50 @@ def _ensure_db_id(self, obj, is_type=False, refs=0):
self._handle_tracked_type(obj, obj_id)
return obj_id

def _maybe_capture_value(self, obj, obj_id):
'''
capture a bounded repr of obj into object_repr, per self.values tier.

tier 'builtins' only touches exact builtin scalar types (C-level
repr, never runs user code); tier 'repr' additionally runs
reprlib.repr on everything else, which EXECUTES user __repr__
mid-dump -- accepted risk; fork isolation (spawn_dump) bounds the
blast radius to dump quality.
'''
limit = self.value_limit
obj_type = type(obj)
truncated = False
if obj_type in _VALUE_TYPES:
if obj_type is str or obj_type is bytes:
# slice BEFORE repr to bound cost on huge strings; repr adds
# quotes/escapes, so bound the text itself too
text = repr(obj[:limit])
truncated = len(obj) > limit or len(text) > limit
text = text[:limit]
elif obj_type is int:
try:
text = repr(obj)
except ValueError: # 3.11+ int_max_str_digits
text = '<int ~{} bits>'.format(obj.bit_length())
truncated = len(text) > limit
text = text[:limit]
else: # float, complex, bool
text = repr(obj)
elif self.values == 'repr':
try:
text = self._reprlib.repr(obj)
except Exception: # user __repr__ may raise; skip, never fail the dump
return
truncated = len(text) > limit
text = text[:limit]
else:
return
# lone surrogates break sqlite's UTF-8 encode
text = text.encode('utf-8', 'backslashreplace').decode('utf-8')
self.execute(
"INSERT INTO object_repr (object, repr, truncated, redacted) VALUES (?, ?, ?, 0)",
(obj_id, text, int(truncated)))

def _handle_tracked_type(self, obj, obj_id):
'''
after creating the row in the object table, handle creating another
Expand Down Expand Up @@ -554,6 +611,65 @@ def add_all(self):
(db_id, self._ensure_db_id(referent, refs=1)))
self.ignore_ids.remove(id(sys._getframe()))

def redact_sensitive_values(self):
'''
replace object_repr rows for values sitting under sensitive-looking
edge labels with a length-bucketed placeholder. key objects (the
string 'password') stay readable -- the key is the clue, the value
is the secret. an object aliased under both a sensitive and an
innocent label is redacted: one sensitive inbound edge wins.
'''
sensitive_dsts = set()
key_dsts = [] # (key_obj_id, dst) pairs from dict-key edges
for dst, ref in self.conn.execute("SELECT dst, ref FROM reference"):
if not isinstance(ref, str): # list-index edges from enumerate
continue
if ref.startswith('@'): # dict-key edge, format '@<key_obj_id>'
key_dsts.append((int(ref[1:]), dst))
continue
ref_lower = ref.lower()
if any(p in ref_lower for p in _SENSITIVE_KEY_PATTERNS):
sensitive_dsts.add(dst)
# dict keys: a key whose captured repr text matches the patterns
# marks its dst sensitive; keys with no captured value can't match
key_text = {}
key_ids = list({key_obj_id for key_obj_id, _ in key_dsts})
for i in range(0, len(key_ids), 800):
chunk = key_ids[i:i + 800]
qs = ', '.join(['?'] * len(chunk))
for obj, text in self.conn.execute(
"SELECT object, repr FROM object_repr WHERE object IN ({})".format(qs), chunk):
key_text[obj] = text.lower()
sensitive_key_obj_ids = set()
for key_obj_id, dst in key_dsts:
text = key_text.get(key_obj_id)
if text is not None and any(p in text for p in _SENSITIVE_KEY_PATTERNS):
sensitive_dsts.add(dst)
sensitive_key_obj_ids.add(key_obj_id)
# keys themselves stay readable (even when aliased as a value
# elsewhere, e.g. CPython 3.14 interned-string self-mapping dicts)
sensitive_dsts -= sensitive_key_obj_ids
# replace captured rows; length basis is object.len when present
# (true pre-truncation length for str/bytes), else the repr length
updates = []
dst_ids = list(sensitive_dsts)
for i in range(0, len(dst_ids), 800):
chunk = dst_ids[i:i + 800]
qs = ', '.join(['?'] * len(chunk))
for obj, text, length in self.conn.execute(
"""
SELECT object_repr.object, object_repr.repr, object.len
FROM object_repr JOIN object ON object_repr.object = object.id
WHERE object_repr.object IN ({})
""".format(qs), chunk):
n = length if length is not None else len(text)
updates.append(('<redacted len<={}>'.format(_len_bucket(n)), obj))
if updates:
self.executemany(
"UPDATE object_repr SET repr = ?, truncated = 0, redacted = 1 WHERE object = ?",
updates)
self.conn.commit()

def finish(self):
self.conn.execute(
"UPDATE meta SET duration_s = ?",
Expand All @@ -577,17 +693,47 @@ def finish(self):
_GC_COMPLETE_TYPES = frozenset([
dict, list, tuple, set, frozenset, collections.deque, collections.defaultdict])

# exact types only: C-level repr, never runs user code (subclasses excluded)
_VALUE_TYPES = (str, bytes, int, float, complex, bool)

_SENSITIVE_KEY_PATTERNS = (
# merged from ffcore.redact_sensitive_data defaults and Sentry
# server-side scrubbing defaults
'key', 'secret', 'passphrase', 'password', 'passwd', 'pwd',
'token', 'auth', 'credential', 'session', 'private',
)

def dump_graph(path, print_info=False, use_gc=False):

def _len_bucket(n):
'''smallest bucket >= n from the series 100, 500, 1000, 5000, 10000, ...'''
bucket, step_five = 100, True
while bucket < n:
bucket *= 5 if step_five else 2
step_five = not step_five
return bucket


def dump_graph(path, print_info=False, use_gc=False, values='none', value_limit=128, redact=True):
'''
dump a collection db to path;
the collection db is designed to be small
and write fast, so it needs post-processing
to e.g. add indices and compute values
before analysis

values -- value-capture tier: 'none' (default; no object values stored),
'builtins' (bounded reprs of exact builtin scalars: str, bytes, int,
float, complex, bool), or 'repr' (builtins plus reprlib.repr of
everything else; runs user __repr__ mid-dump -- prefer spawn_dump for
fork isolation)
value_limit -- max stored repr length per object (default 128)
redact -- when capturing values, replace values under sensitive-looking
keys ('secret', 'token', ...) with a length-bucketed placeholder
(default True)
'''
start = time.time()
_Writer.write_to_path(path, use_gc=use_gc)
_Writer.write_to_path(path, use_gc=use_gc, values=values,
value_limit=value_limit, redact=redact)
if print_info:
duration = time.time() - start
memory = _get_memory_mb()
Expand All @@ -609,7 +755,7 @@ def dump_graph(path, print_info=False, use_gc=False):
return


def spawn_dump(path, print_info=False, use_gc=False):
def spawn_dump(path, print_info=False, use_gc=False, values='none', value_limit=128, redact=True):
if not hasattr(os, 'fork'):
raise NotImplementedError('spawn_dump() requires os.fork() support')

Expand All @@ -618,7 +764,8 @@ def spawn_dump(path, print_info=False, use_gc=False):
return pid

try:
dump_graph(path, print_info=print_info, use_gc=use_gc)
dump_graph(path, print_info=print_info, use_gc=use_gc,
values=values, value_limit=value_limit, redact=redact)
except BaseException:
os._exit(1)
os._exit(0)
Expand Down
10 changes: 9 additions & 1 deletion objex/schema.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,8 @@
memory_mb INTEGER NOT NULL,
gc_info TEXT NOT NULL,
num_gcd_objects INTEGER NOT NULL,
duration_s REAL
duration_s REAL,
capture_values TEXT -- value-capture tier used for this dump ('none', 'builtins', 'repr')
);

CREATE TABLE object (
Expand All @@ -27,6 +28,13 @@
is_gc_tracked INTEGER NOT NULL
);

CREATE TABLE object_repr (
object INTEGER PRIMARY KEY, -- object id
repr TEXT NOT NULL, -- truncated repr text, or redaction placeholder
truncated INTEGER NOT NULL, -- 1 if cut at value_limit
redacted INTEGER NOT NULL -- 1 if replaced by the sensitive-key pass
);

CREATE TABLE pytype (
id INTEGER PRIMARY KEY,
object INTEGER NOT NULL,
Expand Down
1 change: 1 addition & 0 deletions objex/web.py
Original file line number Diff line number Diff line change
Expand Up @@ -407,6 +407,7 @@
<dt>Size</dt><dd>${obj.size}</dd>
<dt>Refcount</dt><dd>${escapeHtml(obj.refcount_display ?? String(obj.refcount))}</dd>
<dt>Len</dt><dd>${obj.len ?? ''}</dd>
${obj.repr != null ? `<dt>Repr</dt><dd>${escapeHtml(obj.repr)}</dd>` : ''}
</dl>
`;
}
Expand Down
Loading
Loading