| pandas — USER GUIDE (Android Python STB) | |
| ========================================= | |
| Generated by RIMI | |
| Version: pandas 2.3.3 | |
| Python: 3.12.14 | |
| WHAT IS PANDAS? | |
| --------------- | |
| pandas is the most popular Python library for data analysis and manipulation. | |
| It provides fast, expressive DataFrames (tabular data) and Series (1D data) | |
| that make working with structured data easy and intuitive. | |
| In PythonSTB, pandas is used for: | |
| - EPG (Electronic Program Guide) data parsing and querying | |
| - Channel list management and filtering | |
| - Playlist data transformation | |
| - Analytics and statistics | |
| - CSV/JSON data processing | |
| QUICK START | |
| ----------- | |
| import pandas as pd | |
| # Create a DataFrame | |
| df = pd.DataFrame({ | |
| "channel": ["BBC One", "CNN", "Sky News"], | |
| "category": ["entertainment", "news", "news"], | |
| "rating": [4.5, 4.2, 4.0] | |
| }) | |
| # Filter | |
| news = df[df["category"] == "news"] | |
| # Sort | |
| top = df.sort_values("rating", ascending=False) | |
| # Save / Load | |
| df.to_csv("channels.csv", index=False) | |
| df = pd.read_csv("channels.csv") | |
| CORE CONCEPTS | |
| ------------- | |
| 1. DataFrame: 2D table (like a spreadsheet or SQL table) | |
| df = pd.DataFrame({"col1": [1,2,3], "col2": ["a","b","c"]}) | |
| 2. Series: 1D column or row | |
| s = df["col1"] | |
| 3. Indexing: | |
| df.loc[row_label, col_label] # label-based | |
| df.iloc[row_int, col_int] # position-based | |
| df[df["col"] > value] # boolean mask | |
| 4. GroupBy: | |
| df.groupby("category")["rating"].mean() | |
| 5. Merge/Join: | |
| pd.merge(df1, df2, on="key") | |
| COMMON OPERATIONS | |
| ----------------- | |
| # Filtering | |
| df[df["rating"] > 4.0] | |
| df.query("rating > 4.0") | |
| df.nlargest(5, "rating") | |
| # Aggregation | |
| df.groupby("category").agg({"rating": "mean", "channel": "count"}) | |
| # Transform | |
| df["normalized"] = (df["rating"] - df["rating"].min()) / (df["rating"].max() - df["rating"].min()) | |
| # Pivot | |
| pd.pivot_table(df, values="rating", index="category", aggfunc="mean") | |
| # Time series | |
| df["date"] = pd.to_datetime(df["date"]) | |
| df.set_index("date").resample("D").mean() | |
| FILE I/O | |
| -------- | |
| # CSV | |
| df.to_csv("data.csv", index=False) | |
| df = pd.read_csv("data.csv") | |
| df = pd.read_csv("data.csv", parse_dates=["date"]) | |
| # JSON | |
| df.to_json("data.json", orient="records") | |
| df = pd.read_json("data.json", orient="records") | |
| # Excel (requires openpyxl) | |
| df.to_excel("data.xlsx", index=False) | |
| df = pd.read_excel("data.xlsx") | |
| EPG DATA EXAMPLE | |
| ---------------- | |
| import pandas as pd | |
| from lxml import etree | |
| def parse_epg(xml_content): | |
| root = etree.fromstring(xml_content.encode()) | |
| rows = [] | |
| for prog in root.findall(".//programme"): | |
| rows.append({ | |
| "channel": prog.get("channel"), | |
| "start": pd.to_datetime(prog.get("start"), format="%Y%m%d%H%M%S %z"), | |
| "stop": pd.to_datetime(prog.get("stop"), format="%Y%m%d%H%M%S %z"), | |
| "title": prog.findtext("title", ""), | |
| }) | |
| return pd.DataFrame(rows) | |
| # Query: what's on now? | |
| now = pd.Timestamp.now(tz="UTC") | |
| on_now = epg[(epg["start"] <= now) & (epg["stop"] > now)] | |
| # Query: shows longer than 1 hour | |
| long_shows = epg[(epg["stop"] - epg["start"]) > pd.Timedelta(hours=1)] | |
| TIPS FOR ANDROID | |
| ---------------- | |
| - pandas on Android is compiled with norelro + 16KB page alignment | |
| - Full wheel bundles numpy — no separate numpy install needed | |
| - Use zipfile-based installer (pip doesn't work from run-as) | |
| - pandas + numpy together use ~15MB installed | |
| - All DataFrame operations work the same as desktop Python | |
| TROUBLESHOOTING | |
| --------------- | |
| # ImportError: numpy required | |
| # Make sure pandas wheel with bundled numpy is installed (Full_Wheel) | |
| # Slow performance | |
| # Use vectorized operations instead of Python loops: | |
| # BAD: for i in range(len(df)): df.loc[i, "new"] = df.loc[i, "old"] * 2 | |
| # GOOD: df["new"] = df["old"] * 2 | |
| # Memory issues with large datasets | |
| # Use chunked reading: | |
| # for chunk in pd.read_csv("big.csv", chunksize=1000): | |
| # process(chunk) | |
| Generated by RIMI | |