Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 23 additions & 2 deletions backend/collectors/collect_partition_statistics.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,8 @@
from collectors.collect_metadata import MetadataFileRecord
from collectors.statistics_collector import HiddenStatisticsMetadata, StatisticsCollector
from collectors.utils import format_partition
from constants import FileType
from constants import PARTITION_STATISTICS_FILE_NOT_READ_WARNING, PARTITION_STATISTICS_READ_LIMIT_WARNING, FileType
from env import Env
from extractors.partition_statistics_extractor import PartitionStatisticsExtractor
from icegraph_logger import logger

Expand Down Expand Up @@ -33,7 +34,27 @@ def _get_pointed_statistics(self, metadata_file: MetadataFileRecord) -> Optional

def _collect_statistics_files(self, metadata_file_to_added_entries: Dict[str, List[dict]]) -> None:
statistics_files = self._build_statistics_files(metadata_file_to_added_entries)
self._collect_partition_summaries(statistics_files)
files_to_read = dict(list(statistics_files.items())[: Env.MAX_PARTITION_STATISTICS_FILES_TO_READ])
if files_to_read:
self._collect_partition_summaries(files_to_read)
self._warn_unread_files([statistics_file for path, statistics_file in statistics_files.items() if path not in files_to_read])

def _warn_unread_files(self, unread_files: List[PartitionStatisticsFileRecord]) -> None:
if not unread_files:
return

for statistics_file in unread_files:
statistics_file.warnings.append(
PARTITION_STATISTICS_FILE_NOT_READ_WARNING.format(max_partition_statistics_files_to_read=Env.MAX_PARTITION_STATISTICS_FILES_TO_READ)
)

self._warnings[f"{self.STATISTICS_KEY}_read_limit"] = [
PARTITION_STATISTICS_READ_LIMIT_WARNING.format(
skipped_files_count=len(unread_files),
total_files_count=len(self._statistics_files),
max_partition_statistics_files_to_read=Env.MAX_PARTITION_STATISTICS_FILES_TO_READ,
)
]

def collect_file(self, metadata_path: str, entry: dict, include_samples: bool = True) -> PartitionStatisticsFileRecord:
statistics_file = self._parse_statistics_entry(metadata_path, entry)
Expand Down
10 changes: 10 additions & 0 deletions backend/constants.py
Original file line number Diff line number Diff line change
Expand Up @@ -80,6 +80,16 @@ class FileType(Enum):
{statistics_name} files that existed before the range may be shown as added by the oldest metadata file in view.
""")

PARTITION_STATISTICS_READ_LIMIT_WARNING = inspect.cleandoc("""
Showing partial partition statistics! {skipped_files_count} of {total_files_count} partition statistics files were not read because the read limit of {max_partition_statistics_files_to_read} was reached.

Only the newest files were read. Older files show only the fields from the metadata file.
""")

PARTITION_STATISTICS_FILE_NOT_READ_WARNING = inspect.cleandoc("""
The partition statistics file was not read because the limit of {max_partition_statistics_files_to_read} partition statistics files to read was reached.
""")

TABLE_STATISTICS_COLLECTION_ERROR = "Failed to read the table statistics entries, so table statistics files are not shown: {error}"

STATISTICS_ATTRIBUTION_WARNING = inspect.cleandoc("""
Expand Down
3 changes: 3 additions & 0 deletions backend/env.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,9 @@ class Env:
# Maximum number of partitions sampled from each partition statistics file.
MAX_PARTITION_STATISTICS_ROWS: int = int(os.getenv("MAX_PARTITION_STATISTICS_ROWS", "10"))

# Maximum number of partition statistics files read per graph, newest first.
MAX_PARTITION_STATISTICS_FILES_TO_READ: int = int(os.getenv("MAX_PARTITION_STATISTICS_FILES_TO_READ", "50"))

# Cache lifetime for the table selection endpoint.
TABLE_LIST_CACHE_TTL_SECONDS: int = int(os.getenv("TABLE_LIST_CACHE_TTL_SECONDS", "60"))

Expand Down
2 changes: 1 addition & 1 deletion frontend/src/features/docs/content/graph-view.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@ The Graph view shows all Iceberg metadata objects in your selected range as a di

- **Table statistics** - Puffin file with column statistics, such as distinct value counts, for one snapshot. Drawn in amber left of the metadata files and linked from the metadata file that added it. IceGraph shows what the metadata file says about it, without opening it

- **Partition statistics** - Parquet file with per-partition counts, such as records, files, and deletes, for one snapshot. Drawn in silver left of the metadata files and linked from the metadata file that added it. IceGraph opens it and shows how the data is spread across partitions, plus a sample of the most recently updated partitions. The distribution caption **Per partition across all partitions** means it covers every partition in the statistics file. The details show `partitions_with_deletes` as a separate count, treating missing or null delete-file counts as zero. Samples use a shared set of fields across the collected statistics files
- **Partition statistics** - Parquet file with per-partition counts, such as records, files, and deletes, for one snapshot. Drawn in silver left of the metadata files and linked from the metadata file that added it. IceGraph opens it and shows how the data is spread across partitions, plus a sample of the most recently updated partitions. The distribution caption **Per partition across all partitions** means it covers every partition in the statistics file. The details show `partitions_with_deletes` as a separate count, treating missing or null delete-file counts as zero. Samples use a shared set of fields across the collected statistics files. Only the newest files, up to a configured limit, are opened; older files show only their metadata fields and a warning

- **Catalog** - not a file. It is the entry point to the table: the catalog is what points to the current main metadata file, and every read of the table starts there. It is drawn larger and highlighted, labelled with the table name, and sits at the top of the metadata column, with a downward arrow to the main metadata file. Its details show the table properties that do not change between updates: name, UUID, location and format version

Expand Down
4 changes: 3 additions & 1 deletion frontend/src/pages/GraphPage.jsx
Original file line number Diff line number Diff line change
Expand Up @@ -156,7 +156,7 @@
nodes: rawNodes,
edges: rawEdges,
metadata,
errors,

Check warning on line 159 in frontend/src/pages/GraphPage.jsx

View workflow job for this annotation

GitHub Actions / frontend

'errors' is assigned a value but never used
} = useTableGraphData();

const search = useSearch({ strict: false });
Expand Down Expand Up @@ -343,7 +343,7 @@
stickyScrollTargetRef.current = 0;
if (stickyPanelRef.current) stickyPanelRef.current.scrollTop = 0;
},
[selectNodeInHistory],

Check warning on line 346 in frontend/src/pages/GraphPage.jsx

View workflow job for this annotation

GitHub Actions / frontend

React Hook useCallback has a missing dependency: 'setStickyNode'. Either include it or remove the dependency array
);

const closeStickyPanel = useCallback(() => {
Expand Down Expand Up @@ -539,9 +539,9 @@
unbindSticky();
unbindPopup();
if (stickyScrollRafRef.current)
cancelAnimationFrame(stickyScrollRafRef.current);

Check warning on line 542 in frontend/src/pages/GraphPage.jsx

View workflow job for this annotation

GitHub Actions / frontend

The ref value 'stickyScrollRafRef.current' will likely have changed by the time this effect cleanup function runs. If this ref points to a node rendered by React, copy 'stickyScrollRafRef.current' to a variable inside the effect, and use that variable in the cleanup function
if (popupScrollRafRef.current)
cancelAnimationFrame(popupScrollRafRef.current);

Check warning on line 544 in frontend/src/pages/GraphPage.jsx

View workflow job for this annotation

GitHub Actions / frontend

The ref value 'popupScrollRafRef.current' will likely have changed by the time this effect cleanup function runs. If this ref points to a node rendered by React, copy 'popupScrollRafRef.current' to a variable inside the effect, and use that variable in the cleanup function
};
}, [navigateTo, resetZoom, resetView, closeStickyPanel]);

Expand Down Expand Up @@ -593,7 +593,7 @@
} else {
setTimeout(() => resetView(), 100);
}
}, [

Check warning on line 596 in frontend/src/pages/GraphPage.jsx

View workflow job for this annotation

GitHub Actions / frontend

React Hook useEffect has a missing dependency: 'setStickyNode'. Either include it or remove the dependency array
graphData,
graphSelectionFromHistory,
selectNodeIdFromSearch,
Expand Down Expand Up @@ -661,7 +661,7 @@
setStickyNode(node);
selectNodeInHistory(node.id);
},
[graphData, selectNodeInHistory],

Check warning on line 664 in frontend/src/pages/GraphPage.jsx

View workflow job for this annotation

GitHub Actions / frontend

React Hook useCallback has a missing dependency: 'setStickyNode'. Either include it or remove the dependency array
);

const paintNode = useCallback(
Expand Down Expand Up @@ -1037,6 +1037,7 @@
/>
))}
{stickyNode.details.type === FileType.PARTITION_STATISTICS &&
stickyNode.details.partitions_count != null &&
stickyNode.details.partition_distribution && (
<PartitionDistributionTable
partitionDistribution={
Expand All @@ -1045,9 +1046,10 @@
/>
)}
{stickyNode.details.type === FileType.PARTITION_STATISTICS &&
stickyNode.details.partitions_count != null &&
Array.isArray(stickyNode.details.sampled_partitions) && (
<PartitionStatisticsTable
partitionsCount={stickyNode.details.partitions_count ?? null}
partitionsCount={stickyNode.details.partitions_count}
sampledPartitions={stickyNode.details.sampled_partitions}
/>
)}
Expand Down
Loading