Section 10: Programming

🐍 MongoDB with Python

Master PyMongo - Complete guide from basics to production-ready applications

βš™οΈ Setup & Installation

Install PyMongo

# Using pip
pip install pymongo

# Using poetry
poetry add pymongo

# With specific version
pip install pymongo==4.6.1

# Verify installation
python -c "import pymongo; print(pymongo.version)"

Project Structure

my_app/
β”œβ”€β”€ config/
β”‚   └── database.py      # Connection settings
β”œβ”€β”€ models/
β”‚   β”œβ”€β”€ user.py          # User model
β”‚   └── product.py       # Product model
β”œβ”€β”€ repositories/
β”‚   β”œβ”€β”€ user_repo.py     # User CRUD
β”‚   └── product_repo.py  # Product CRUD
β”œβ”€β”€ services/
β”‚   └── user_service.py  # Business logic
β”œβ”€β”€ .env                 # Environment variables
β”œβ”€β”€ requirements.txt     # Dependencies
└── main.py             # Entry point

πŸ”— Database Connection

Basic Connection

from pymongo import MongoClient

# Connect to localhost
client = MongoClient('mongodb://localhost:27017/')

# Get database
db = client['myapp_db']

# Get collection
users = db['users']

print("Connected to MongoDB!")

Connection with Authentication

from pymongo import MongoClient

# Connection string with auth
uri = "mongodb://username:password@localhost:27017/"
client = MongoClient(uri)

# Or using parameters
client = MongoClient(
    host='localhost',
    port=27017,
    username='admin',
    password='secretpass',
    authSource='admin'
)

db = client['production_db']

MongoDB Atlas Connection

from pymongo import MongoClient
import os
from dotenv import load_dotenv

load_dotenv()

# Atlas connection string
MONGO_URI = os.getenv('MONGO_URI')
client = MongoClient(MONGO_URI)

db = client['cloud_app']
print(f"Connected to database: {db.name}")

Connection Pooling (Production)

from pymongo import MongoClient

client = MongoClient(
    'mongodb://localhost:27017/',
    maxPoolSize=50,        # Max connections in pool
    minPoolSize=10,        # Min connections
    maxIdleTimeMS=45000,   # Close idle connections after 45s
    serverSelectionTimeoutMS=5000  # Timeout for server selection
)

# Singleton pattern for production
class Database:
    _instance = None
    
    @classmethod
    def get_db(cls):
        if cls._instance is None:
            client = MongoClient('mongodb://localhost:27017/')
            cls._instance = client['myapp']
        return cls._instance

# Usage
db = Database.get_db()

✏️ CRUD Operations

Create (Insert)

from pymongo import MongoClient
from datetime import datetime

client = MongoClient('mongodb://localhost:27017/')
db = client['myapp']
users = db['users']

# Insert one document
user = {
    'name': 'Alice Johnson',
    'email': 'alice@example.com',
    'age': 28,
    'created_at': datetime.utcnow(),
    'tags': ['python', 'mongodb', 'data-science']
}

result = users.insert_one(user)
print(f"Inserted ID: {result.inserted_id}")

# Insert multiple documents
new_users = [
    {'name': 'Bob Smith', 'email': 'bob@example.com', 'age': 32},
    {'name': 'Charlie Brown', 'email': 'charlie@example.com', 'age': 25}
]

result = users.insert_many(new_users)
print(f"Inserted {len(result.inserted_ids)} documents")

Read (Query)

# Find one document
user = users.find_one({'email': 'alice@example.com'})
print(user)

# Find all documents
all_users = users.find()
for user in all_users:
    print(user['name'])

# Find with filter
young_users = users.find({'age': {'$lt': 30}})
for user in young_users:
    print(f"{user['name']} - {user['age']} years old")

# Find with projection (select specific fields)
names_only = users.find(
    {},
    {'name': 1, 'email': 1, '_id': 0}
)
for user in names_only:
    print(user)

# Count documents
count = users.count_documents({'age': {'$gte': 25}})
print(f"Users 25+: {count}")

Update

# Update one document
result = users.update_one(
    {'email': 'alice@example.com'},
    {'$set': {'age': 29, 'updated_at': datetime.utcnow()}}
)
print(f"Modified {result.modified_count} document(s)")

# Update multiple documents
result = users.update_many(
    {'age': {'$lt': 30}},
    {'$set': {'category': 'young_professional'}}
)
print(f"Updated {result.modified_count} users")

# Increment a field
users.update_one(
    {'email': 'alice@example.com'},
    {'$inc': {'login_count': 1}}
)

# Add to array
users.update_one(
    {'email': 'alice@example.com'},
    {'$push': {'tags': 'machine-learning'}}
)

# Upsert (insert if not exists)
result = users.update_one(
    {'email': 'new@example.com'},
    {'$set': {'name': 'New User', 'age': 30}},
    upsert=True
)
print(f"Upserted: {result.upserted_id}")

Delete

# Delete one document
result = users.delete_one({'email': 'old@example.com'})
print(f"Deleted {result.deleted_count} document")

# Delete multiple documents
result = users.delete_many({'age': {'$gt': 60}})
print(f"Deleted {result.deleted_count} users")

# Delete all documents (careful!)
# result = users.delete_many({})

# Drop entire collection (very careful!)
# users.drop()

πŸ” Advanced Query Operations

Comparison Operators

# Greater than, less than
users.find({'age': {'$gt': 25, '$lt': 40}})

# In array
users.find({'status': {'$in': ['active', 'pending']}})

# Not in array
users.find({'status': {'$nin': ['banned', 'deleted']}})

# Exists
users.find({'premium': {'$exists': True}})

# Type checking
users.find({'age': {'$type': 'int'}})

Logical Operators

# AND (implicit)
users.find({'age': {'$gte': 25}, 'status': 'active'})

# OR
users.find({
    '$or': [
        {'age': {'$lt': 25}},
        {'status': 'premium'}
    ]
})

# NOT
users.find({'age': {'$not': {'$gt': 30}}})

# NOR
users.find({
    '$nor': [
        {'age': {'$lt': 18}},
        {'status': 'banned'}
    ]
})

Array Queries

# Array contains value
users.find({'tags': 'python'})

# All elements match
users.find({'tags': {'$all': ['python', 'mongodb']}})

# Array size
users.find({'tags': {'$size': 3}})

# Element match
users.find({
    'orders': {
        '$elemMatch': {
            'status': 'shipped',
            'total': {'$gt': 100}
        }
    }
})

Text Search

# Create text index first
users.create_index([('name', 'text'), ('bio', 'text')])

# Text search
results = users.find({'$text': {'$search': 'python developer'}})
for user in results:
    print(user['name'])

Sorting, Limiting, Skipping

# Sort ascending
users.find().sort('age', 1)

# Sort descending
users.find().sort('created_at', -1)

# Multiple sort fields
users.find().sort([('age', -1), ('name', 1)])

# Limit results
users.find().limit(10)

# Skip documents (pagination)
page = 2
page_size = 10
users.find().skip((page - 1) * page_size).limit(page_size)

# Combine all
users.find({'status': 'active'}) \
    .sort('created_at', -1) \
    .skip(20) \
    .limit(10)

πŸ“Š Aggregation Pipeline

Basic Aggregation

# Group and count
pipeline = [
    {'$group': {
        '_id': '$status',
        'count': {'$sum': 1}
    }}
]
results = users.aggregate(pipeline)
for result in results:
    print(f"{result['_id']}: {result['count']}")

# Average age by category
pipeline = [
    {'$group': {
        '_id': '$category',
        'avg_age': {'$avg': '$age'},
        'total': {'$sum': 1}
    }}
]
results = users.aggregate(pipeline)

Complex Aggregation

# Multi-stage pipeline
pipeline = [
    # Stage 1: Filter
    {'$match': {'age': {'$gte': 25}}},
    
    # Stage 2: Group
    {'$group': {
        '_id': '$city',
        'count': {'$sum': 1},
        'avg_age': {'$avg': '$age'},
        'total_orders': {'$sum': '$order_count'}
    }},
    
    # Stage 3: Sort
    {'$sort': {'count': -1}},
    
    # Stage 4: Limit
    {'$limit': 5},
    
    # Stage 5: Project (reshape)
    {'$project': {
        'city': '$_id',
        'users': '$count',
        'average_age': {'$round': ['$avg_age', 1]},
        '_id': 0
    }}
]

results = users.aggregate(pipeline)
for city_data in results:
    print(city_data)

Lookup (Join)

# Join users with orders
pipeline = [
    {'$lookup': {
        'from': 'orders',
        'localField': '_id',
        'foreignField': 'user_id',
        'as': 'user_orders'
    }},
    {'$project': {
        'name': 1,
        'email': 1,
        'order_count': {'$size': '$user_orders'}
    }}
]

results = users.aggregate(pipeline)

πŸš€ Indexing

# Create single field index
users.create_index('email', unique=True)

# Create compound index
users.create_index([('age', 1), ('status', 1)])

# Create text index
users.create_index([('name', 'text'), ('bio', 'text')])

# Create TTL index (auto-delete after time)
from datetime import timedelta
sessions.create_index(
    'created_at',
    expireAfterSeconds=3600  # Delete after 1 hour
)

# List all indexes
indexes = users.list_indexes()
for index in indexes:
    print(index)

# Drop index
users.drop_index('email_1')

πŸ’³ Transactions

from pymongo import MongoClient

client = MongoClient('mongodb://localhost:27017/')
db = client['banking']

# Transfer money between accounts
def transfer_money(from_account, to_account, amount):
    with client.start_session() as session:
        with session.start_transaction():
            try:
                # Debit from account
                db.accounts.update_one(
                    {'account_id': from_account},
                    {'$inc': {'balance': -amount}},
                    session=session
                )
                
                # Credit to account
                db.accounts.update_one(
                    {'account_id': to_account},
                    {'$inc': {'balance': amount}},
                    session=session
                )
                
                # Log transaction
                db.transactions.insert_one({
                    'from': from_account,
                    'to': to_account,
                    'amount': amount,
                    'timestamp': datetime.utcnow()
                }, session=session)
                
                print("Transfer successful!")
                
            except Exception as e:
                print(f"Transfer failed: {e}")
                session.abort_transaction()
                raise

# Execute transfer
transfer_money('A123', 'B456', 100.00)

⭐ Best Practices

Production Best Practices:

  1. Use Connection Pooling: Singleton pattern for DB connection
  2. Index Your Queries: Create indexes on frequently queried fields
  3. Use Projections: Only fetch fields you need
  4. Batch Operations: Use insert_many, update_many, delete_many
  5. Error Handling: Wrap operations in try-except blocks
  6. Environment Variables: Store credentials in .env files
  7. Connection Limits: Set maxPoolSize appropriately
  8. Close Connections: Use context managers
  9. Validate Data: Use Pydantic models for validation
  10. Monitor Performance: Use explain() to analyze queries

πŸ—οΈ Complete Flask + MongoDB Project

Project Structure

# app.py - Flask application
from flask import Flask, request, jsonify
from pymongo import MongoClient
from datetime import datetime
import os

app = Flask(__name__)

# Database connection
client = MongoClient(os.getenv('MONGO_URI', 'mongodb://localhost:27017/'))
db = client['blog_app']
posts = db['posts']

# Create post
@app.route('/posts', methods=['POST'])
def create_post():
    data = request.json
    post = {
        'title': data['title'],
        'content': data['content'],
        'author': data['author'],
        'tags': data.get('tags', []),
        'created_at': datetime.utcnow(),
        'views': 0
    }
    result = posts.insert_one(post)
    return jsonify({'id': str(result.inserted_id)}), 201

# Get all posts
@app.route('/posts', methods=['GET'])
def get_posts():
    page = int(request.args.get('page', 1))
    limit = int(request.args.get('limit', 10))
    
    cursor = posts.find() \
        .sort('created_at', -1) \
        .skip((page - 1) * limit) \
        .limit(limit)
    
    result = []
    for post in cursor:
        post['_id'] = str(post['_id'])
        result.append(post)
    
    return jsonify(result)

# Get post by ID
@app.route('/posts/', methods=['GET'])
def get_post(post_id):
    from bson import ObjectId
    
    post = posts.find_one({'_id': ObjectId(post_id)})
    if post:
        post['_id'] = str(post['_id'])
        # Increment view count
        posts.update_one(
            {'_id': ObjectId(post_id)},
            {'$inc': {'views': 1}}
        )
        return jsonify(post)
    return jsonify({'error': 'Not found'}), 404

# Update post
@app.route('/posts/', methods=['PUT'])
def update_post(post_id):
    from bson import ObjectId
    
    data = request.json
    result = posts.update_one(
        {'_id': ObjectId(post_id)},
        {'$set': {
            'title': data.get('title'),
            'content': data.get('content'),
            'updated_at': datetime.utcnow()
        }}
    )
    
    if result.modified_count:
        return jsonify({'message': 'Updated'})
    return jsonify({'error': 'Not found'}), 404

# Delete post
@app.route('/posts/', methods=['DELETE'])
def delete_post(post_id):
    from bson import ObjectId
    
    result = posts.delete_one({'_id': ObjectId(post_id)})
    if result.deleted_count:
        return jsonify({'message': 'Deleted'})
    return jsonify({'error': 'Not found'}), 404

if __name__ == '__main__':
    app.run(debug=True)

πŸ’Ό Interview Questions & Answers

Q1 How do you connect to MongoDB in Python? β–Ό

Using PyMongo library:

from pymongo import MongoClient

# Basic connection
client = MongoClient('mongodb://localhost:27017/')
db = client['database_name']
collection = db['collection_name']

# With authentication
client = MongoClient(
    'mongodb://username:password@host:port/',
    authSource='admin'
)

Best practice: Use environment variables for credentials and implement singleton pattern for production.

Q2 What's the difference between insert_one() and insert_many()? β–Ό

insert_one(): Inserts a single document

result = collection.insert_one({'name': 'Alice'})
print(result.inserted_id)  # ObjectId of inserted doc

insert_many(): Inserts multiple documents at once (more efficient)

docs = [{'name': 'Bob'}, {'name': 'Charlie'}]
result = collection.insert_many(docs)
print(result.inserted_ids)  # List of ObjectIds

Performance: insert_many() is much faster for bulk inserts (single round trip to DB)

Q3 How do you implement pagination in PyMongo? β–Ό

Using skip() and limit():

def get_paginated_results(page=1, page_size=10):
    skip_count = (page - 1) * page_size
    
    cursor = collection.find() \
        .skip(skip_count) \
        .limit(page_size) \
        .sort('created_at', -1)
    
    results = list(cursor)
    total = collection.count_documents({})
    
    return {
        'results': results,
        'page': page,
        'total_pages': (total + page_size - 1) // page_size,
        'total': total
    }

Note: For large offsets, use cursor-based pagination (query with last_id) instead of skip()

Q4 How do you handle transactions in PyMongo? β–Ό

Using sessions and transactions:

with client.start_session() as session:
    with session.start_transaction():
        try:
            # Operation 1
            collection1.update_one(
                {'_id': id1},
                {'$inc': {'balance': -100}},
                session=session
            )
            
            # Operation 2
            collection2.update_one(
                {'_id': id2},
                {'$inc': {'balance': 100}},
                session=session
            )
            
            # Auto-commits if no exception
        except Exception as e:
            # Auto-aborts on exception
            raise

Requirements: Replica set or sharded cluster (not standalone MongoDB)

Q5 What are PyMongo best practices for production? β–Ό

Production best practices:

  1. Connection pooling: Use singleton pattern, set maxPoolSize
  2. Indexes: Create indexes on frequently queried fields
  3. Projections: Only fetch needed fields to reduce bandwidth
  4. Batch operations: Use insert_many, update_many for bulk ops
  5. Error handling: Wrap operations in try-except blocks
  6. Environment variables: Never hardcode credentials
  7. Timeouts: Set serverSelectionTimeoutMS
  8. Monitoring: Use explain() to analyze query performance
  9. Validation: Use Pydantic or schemas for data validation
  10. Connection cleanup: Close connections properly